
#!/usr/bin/env python3
import requests, sqlite3, re, time, hashlib
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '\u96c5\u5b89\u7ecf\u5f00\u533a\u901a\u77e5\u516c\u544a'
BASE_URL = 'https://jkq.yaan.gov.cn'
LIST_URL = BASE_URL + '/xinwen/list/ed38700d-1b4c-4176-a0f7-e9423aa33291.html'
CUTOFF_DATE = '2023-06-01'

HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
session = requests.Session()
session.headers.update(HEADERS)

def make_int_id(xid):
    try:
        return int(xid)
    except ValueError:
        h = hashlib.md5(xid.encode()).hexdigest()[:15]
        return int(h, 16) % (2**63)

def fetch_list(page=1):
    url = LIST_URL if page == 1 else f'{LIST_URL}?page={page}'
    r = session.get(url, timeout=30)
    html = r.content.decode('utf-8', errors='replace')
    items = re.findall(r'<a[^>]*title="([^"]*)"[^>]*href="(/xinwen/show/[^"]+)"[^>]*>.*?<span>([^<]+)</span>', html, re.DOTALL)
    result = []
    for title, href, date in items:
        if title and len(title) > 10 and '/xinwen/show/' in href:
            result.append({'title': title, 'href': href, 'date': date})
    return result

def fetch_detail(href):
    r = session.get(f'{BASE_URL}{href}', timeout=30)
    html = r.content.decode('utf-8', errors='replace')
    
    pub_date = ''
    dm = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if dm:
        pub_date = dm.group(1)
    
    content_html = ''
    m = re.search(r'<div class="xqing-web">(.*?)</div>', html, re.DOTALL)
    if m:
        h = m.group(1)
        h = re.sub(r'<script[^>]*>.*?</script>', '', h, flags=re.DOTALL|re.I)
        h = re.sub(r'<style[^>]*>.*?</style>', '', h, flags=re.DOTALL|re.I)
        text_only = re.sub(r'<[^>]+>', '', h).strip()
        if len(text_only) >= 20:
            content_html = h.strip()
    
    return content_html, pub_date

def main():
    print(f'=== {SITE_NAME} ===')
    
    all_articles = []
    for pg in range(1, 90):
        try:
            arts = fetch_list(pg)
            if not arts:
                print(f'  Page {pg}: empty, stopping')
                break
            all_articles.extend(arts)
            print(f'  List page {pg}: {len(arts)} articles')
            time.sleep(0.15)
        except Exception as e:
            print(f'  [ERROR] List page {pg}: {e}')
            break
    
    print(f'\nTotal: {len(all_articles)}')
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted, skipped, old, no_content = 0, 0, 0, 0
    start = time.time()
    
    for i, art in enumerate(all_articles):
        title = art['title']
        pub_date = art['date']
        href = art['href']
        
        if pub_date and pub_date < CUTOFF_DATE:
            old += 1
            continue
        
        try:
            content_html, detail_date = fetch_detail(href)
        except Exception as e:
            print(f'  [ERROR] Detail {href}: {e}')
            continue
        
        if detail_date:
            pub_date = detail_date
        if not content_html:
            no_content += 1
        
        int_id = make_int_id(href)
        try:
            c.execute('INSERT OR IGNORE INTO gov_raw(id,site_name,source_url,page_url,title,publish_date,content,date_rank) VALUES(?,?,?,?,?,?,?,?)',
                      (int_id, SITE_NAME,
                       f'{BASE_URL}{href}',
                       f'{BASE_URL}{href}',
                       title, pub_date, content_html,
                       int(pub_date.replace('-','')) if pub_date else 0))
            if c.rowcount > 0: inserted += 1
            else: skipped += 1
        except Exception as e:
            print(f'  [DB ERR] {href}: {e}')
            skipped += 1
        
        if (i+1) % 50 == 0:
            elapsed = time.time() - start
            print(f'  +{inserted}/{(i+1)} (-{skipped}, >3y{old}) [{elapsed:.0f}s]')
    
    conn.commit()
    elapsed = time.time() - start
    print(f'\nNo content: {no_content}')
    print(f'\nSyncing FTS...')
    try:
        c2 = conn.cursor()
        # QC20260926 去掉手写 gov_search 整站删除(抢锁源; FTS 由 gov_raw 触发器维护) 
        # c2.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
        rows = c2.execute("SELECT id, title, site_name, summary, content FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
        c2.executemany("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary, content) VALUES(?,?,?,?,?)", rows)
        conn.commit()
        print(f'  FTS: {len(rows)} records')
    except Exception as e:
        print(f'  FTS error: {e}')
    conn.close()
    print(f'\n=== Done ({elapsed:.0f}s) === +{inserted}, -{skipped}, >3y{old}')

if __name__ == '__main__':
    main()
