
#!/usr/bin/env python3
import requests, sqlite3, re, time, hashlib
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '\u5f00\u9c81\u53bf\u901a\u77e5\u516c\u544a'
BASE_URL = 'http://www.kailu.gov.cn'
LIST_URL = BASE_URL + '/xwzx/tzgggs'
CUTOFF_DATE = '2023-06-01'

HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
session = requests.Session()
session.headers.update(HEADERS)

def make_int_id(xid):
    try:
        return int(xid)
    except ValueError:
        h = hashlib.md5(xid.encode()).hexdigest()[:15]
        return int(h, 16) % (2**63)

def fetch_list(page=0):
    if page == 0:
        url = f'{LIST_URL}/index.html'
    else:
        url = f'{LIST_URL}/index_{page}.html'
    r = session.get(url, timeout=30)
    html = r.content.decode('utf-8', errors='replace')
    articles = []
    links = re.findall(r'<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>', html, re.DOTALL)
    for href, text in links:
        atext = re.sub(r'<[^>]+>', '', text).strip()
        if atext and len(atext) > 10 and href.startswith('./'):
            articles.append({'title': atext, 'href': href[2:]})
    return articles

def fetch_detail(rel_path):
    url = f'{LIST_URL}/{rel_path}'
    r = session.get(url, timeout=30)
    html = r.content.decode('utf-8', errors='replace')
    
    pub_date = ''
    dm = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if dm:
        pub_date = dm.group(1)
    
    # Try multiple content patterns
    content_html = ''
    
    # Pattern 1: TRS_UEDITOR div (most common)
    m = re.search(r'<div[^>]*class="[^"]*TRS_UEDITOR[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if m:
        h = m.group(1)
    else:
        m = re.search(r'<div[^>]*class="[^"]*trs_editor_view[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
        if m:
            h = m.group(1)
        # Pattern 3: em#pareContent
        else:
            pm = re.search(r'<em[^>]*id="pareContent"[^>]*>(.*?)</em>', html, re.DOTALL)
            if pm:
                h = pm.group(1)
            else:
                h = ''
    
    if h:
        h = re.sub(r'<script[^>]*>.*?</script>', '', h, flags=re.DOTALL|re.I)
        h = re.sub(r'<style[^>]*>.*?</style>', '', h, flags=re.DOTALL|re.I)
        content_html = h.strip()
    
    return content_html, pub_date

def main():
    print(f'=== {SITE_NAME} ===')
    
    all_articles = []
    for pg in range(0, 34):
        try:
            arts = fetch_list(pg)
            if not arts:
                break
            all_articles.extend(arts)
            print(f'  List page {pg}: {len(arts)} articles')
            time.sleep(0.15)
        except Exception as e:
            print(f'  [ERROR] List page {pg}: {e}')
    
    print(f'\nTotal: {len(all_articles)}')
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted, skipped, old, no_content = 0, 0, 0, 0
    start = time.time()
    
    for i, art in enumerate(all_articles):
        title = art['title']
        href = art['href']
        
        try:
            content_html, pub_date = fetch_detail(href)
        except Exception as e:
            print(f'  [ERROR] Detail {href}: {e}')
            continue
        
        if not pub_date:
            dm = re.search(r'/(\d{4})(\d{2})/', href)
            if dm:
                pub_date = f'{dm.group(1)}-{dm.group(2)}-01'
        
        if pub_date and pub_date < CUTOFF_DATE:
            old += 1
            continue
        if not content_html:
            no_content += 1
        
        int_id = make_int_id(href)
        try:
            c.execute('INSERT OR IGNORE INTO gov_raw(id,site_name,source_url,page_url,title,publish_date,content,date_rank) VALUES(?,?,?,?,?,?,?,?)',
                      (int_id, SITE_NAME,
                       f'{LIST_URL}/{href}',
                       f'{LIST_URL}/{href}',
                       title, pub_date, content_html,
                       int(pub_date.replace('-','')) if pub_date else 0))
            if c.rowcount > 0: inserted += 1
            else: skipped += 1
        except Exception as e:
            print(f'  [DB ERR] {href}: {e}')
            skipped += 1
        
        if (i+1) % 50 == 0:
            elapsed = time.time() - start
            print(f'  +{inserted}/{(i+1)} (-{skipped}, >3y{old}) [{elapsed:.0f}s]')
    
    conn.commit()
    elapsed = time.time() - start
    print(f'\nNo content: {no_content}/{inserted+skipped}')
    print(f'\nSyncing FTS...')
    try:
        c2 = conn.cursor()
        c2.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
        rows = c2.execute("SELECT id, title, site_name, summary, content FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
        c2.executemany("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary, content) VALUES(?,?,?,?,?)", rows)
        conn.commit()
        print(f'  FTS: {len(rows)} records')
    except Exception as e:
        print(f'  FTS error: {e}')
    conn.close()
    print(f'\n=== Done ({elapsed:.0f}s) === +{inserted}, -{skipped}, >3y{old}')

if __name__ == '__main__':
    main()
