
#!/usr/bin/env python3
import requests, json, sqlite3, re, time, hashlib

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '\u9752\u539f\u533a\u516c\u793a\u516c\u544a'
BASE_URL = 'http://www.qyq.gov.cn'
CUTOFF_DATE = '2023-06-01'

HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
           'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
           'Accept-Language': 'zh-CN,zh;q=0.9'}

session = requests.Session()
session.headers.update(HEADERS)

def build_ajax_data(page):
    return [
        ('ajax_type[0]', '13_xxgk'),
        ('ajax_type[1]', '134709'),
        ('ajax_type[2]', '13'),
        ('ajax_type[3]', 'xxgk'),
        ('ajax_type[4]', 'Y-m-d'),
        ('ajax_type[5]', '50'),
        ('ajax_type[6]', '20'),
        ('ajax_type[7][0]', 'is_top DESC'),
        ('ajax_type[7][1]', 'displayorder DESC'),
        ('ajax_type[7][2]', 'inputtime DESC'),
        ('ajax_type[8]', ''),
        ('is_ds', '1'),
    ]

def fetch_list(page):
    r = session.post(f'{BASE_URL}/api-ajax_list-{page}.html', data=build_ajax_data(page), timeout=30)
    return r.json()

def fetch_detail_with_date(art_id):
    r = session.get(f'{BASE_URL}/xxgk-show-{art_id}.html', timeout=30)
    html = r.content.decode('utf-8', errors='replace')
    
    pub_date = ''
    dm = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if dm:
        pub_date = dm.group(1)
    
    content_html = ''
    m = re.search(r'<div[^>]*class="[^"]*xxgk_content[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        h = m.group(1)
        h = re.sub(r'<script[^>]*>.*?</script>', '', h, flags=re.DOTALL|re.I)
        h = re.sub(r'<style[^>]*>.*?</style>', '', h, flags=re.DOTALL|re.I)
        text_only = re.sub(r'<[^>]+>', '', h).strip()
        
        if len(text_only) >= 20:
            content_html = h.strip()
        elif '<iframe' in h:
            # PDF iframe - generate download link
            ifm = re.search(r'<iframe[^>]*src="([^"]+)"', h)
            if ifm:
                pdf_url = ifm.group(1)
                if not pdf_url.startswith('http'):
                    pdf_url = BASE_URL + pdf_url
                fname = pdf_url.split('/')[-1]
                content_html = f'<div class="article-attachments"><h3>\u9644\u4ef6</h3><ul><li><a href="{pdf_url}" target="_blank" download>{fname}</a></li></ul></div>'
    
    return content_html, pub_date

def make_int_id(xid):
    try:
        return int(xid)
    except ValueError:
        h = hashlib.md5(xid.encode()).hexdigest()[:15]
        return int(h, 16) % (2**63)

def main():
    print(f'=== {SITE_NAME} ===')
    d1 = fetch_list(1)
    total = d1.get('total', len(d1['data']))
    total_pages = (total + 19) // 20
    print(f'Total: {total}, Pages: {total_pages}')
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted, skipped, old, no_content = 0, 0, 0, 0
    start = time.time()
    
    for pg in range(1, min(total_pages, _MAX_PG or total_pages) + 1):
        try:
            d = fetch_list(pg)
        except Exception as e:
            print(f'  [ERROR] List page {pg}: {e}')
            time.sleep(2)
            continue
        arts = d.get('data', [])
        if not arts:
            break
        
        for art in arts:
            title = (art.get('title', '') or '').strip()
            pub_date = (art.get('inputtime', '') or '')[:10]
            if pub_date and pub_date < CUTOFF_DATE:
                old += 1
                continue
            art_id = art.get('id', '')
            url = art.get('url', '') or f'{BASE_URL}/xxgk-show-{art_id}.html'
            if not art_id or not title:
                skipped += 1
                continue
            try:
                content_html, detail_date = fetch_detail_with_date(art_id)
            except Exception as e:
                print(f'  [ERROR] Detail {art_id}: {e}')
                content_html = ''
                detail_date = ''
            if detail_date:
                pub_date = detail_date
            if not content_html:
                no_content += 1
            int_id = make_int_id(art_id)
            try:
                c.execute('INSERT OR IGNORE INTO gov_raw(id,site_name,source_url,page_url,title,publish_date,content,date_rank) VALUES(?,?,?,?,?,?,?,?)',
                          (int_id, SITE_NAME, url, url, title, pub_date, content_html,
                           int(pub_date.replace('-','')) if pub_date else 0))
                if c.rowcount > 0: inserted += 1
                else: skipped += 1
            except Exception as e:
                print(f'  [DB ERR] {art_id}: {e}')
                skipped += 1
        
        elapsed = time.time() - start
        print(f'  Page {pg}/{total_pages}: +{inserted} (-{skipped}, >3y{old}, nc{no_content}) [{elapsed:.0f}s]')
    
    conn.commit()
    elapsed = time.time() - start
    print(f'\nNo content: {no_content}/{inserted+skipped}')
    print(f'\nSyncing FTS...')
    try:
        c2 = conn.cursor()
        # QC20260926 去掉手写 gov_search 整站删除(抢锁源; FTS 由 gov_raw 触发器维护) 
        # c2.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
        rows = c2.execute("SELECT id, title, site_name, summary, content FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
        c2.executemany("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary, content) VALUES(?,?,?,?,?)", rows)
        conn.commit()
        print(f'  FTS: {len(rows)} records')
    except Exception as e:
        print(f'  FTS error: {e}')
    conn.close()
    print(f'\n=== Done ({elapsed:.0f}s) === +{inserted}, -{skipped}, >3y{old}')

if __name__ == '__main__':
    main()
