#!/usr/bin/env python3
"""Fix whitespace cleanup and add summary safety"""
import sys

FILE1 = '/root/gov_crawler/crawl_yuanjiang_eia.py'
FILE2 = '/root/search_app.py'

# Fix 1: Add \\u2002 etc to whitespace cleanup in crawler
with open(FILE1, 'r') as f:
    c1 = f.read()

old1 = "text = re.sub(r'[ \t]+', ' ', text)"
new1 = "text = re.sub(r'[ \t\u2002\u2003\u2005\u00a0]+', ' ', text)"
if old1 in c1:
    c1 = c1.replace(old1, new1)
    with open(FILE1, 'w') as f:
        f.write(c1)
    print("Fix 1: Bad chars added to whitespace cleanup")
else:
    print("Fix 1: Pattern not found")
    # Also try with single quotes
    old1b = "text = re.sub(r'[ \\t]+', ' ', text)"
    if old1b in c1:
        c1 = c1.replace(old1b, new1)
        with open(FILE1, 'w') as f:
            f.write(c1)
        print("Fix 1b: Bad chars added (single quote variant)")
    else:
        print(f"Fix 1: Neither variant found. Lines with '[ \\\\t':")
        for i, line in enumerate(c1.split(chr(10))):
            if '[ \t]' in line or '[ \\t]' in line:
                print(f"  Line {i+1}: {line.rstrip()}")

# Fix 2: Summary safety in search_app - ensure we don't cut in middle of HTML tag
with open(FILE2, 'r') as f:
    c2 = f.read()

# Find where summary is set from content
# Look for content[:500] or summary = content[:500] patterns
old2 = "(content, content[:500],"
new2 = "(content, content[:500].rsplit('<', 1)[0],"
if old2 in c2:
    count = c2.count(old2)
    c2 = c2.replace(old2, new2)
    with open(FILE2, 'w') as f:
        f.write(c2)
    print(f"Fix 2: Summary safety added ({count} occurrences)")
else:
    print("Fix 2: Pattern not found")
    # Show what we have
    for i, line in enumerate(c2.split(chr(10))):
        if 'content[:500]' in line or 'content[:' in line:
            print(f"  Line {i+1}: {line.rstrip()}")
