
#!/usr/bin/env python3
import requests, json, sqlite3, re, time, hashlib
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '\u6f4d\u574a\u6ee8\u6d77\u73af\u8bc4\u516c\u793a'
BASE_URL = 'http://www.wfbinhai.gov.cn'
API_URL = BASE_URL + '/els-service/article'
CUTOFF_DATE = '2023-06-01'
CATAID = '1846716046573178880'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Content-Type': 'application/json',
    'Accept': 'application/json',
}

session = requests.Session()
session.headers.update(HEADERS)

def make_int_id(xxid):
    try:
        return int(xxid)
    except ValueError:
        h = hashlib.md5(xxid.encode()).hexdigest()[:15]
        return int(h, 16) % (2**63)

def fetch_list(page, page_size=15):
    body = {"catas": [CATAID], "dq": "124", "order": "up"}
    resp = session.post(f'{API_URL}/{page}/{page_size}', json=body, timeout=30)
    return resp.json()['data']

def fetch_detail(dq, dwid, xxid):
    url = f'{BASE_URL}/{dq}/{dwid}/{xxid}.html'
    resp = session.get(url, timeout=30)
    html = resp.content.decode('utf-8', errors='replace')
    parts = []
    # Extract main text content
    m = re.search(r'<div class="info-con" id="mainText">(.*?)</div>\s*</div>', html, re.DOTALL)
    if m:
        h = m.group(1)
        h = re.sub(r'<script[^>]*>.*?</script>', '', h, flags=re.DOTALL|re.I)
        h = re.sub(r'<style[^>]*>.*?</style>', '', h, flags=re.DOTALL|re.I)
        h = h.strip()
        if h:
            parts.append(h)
    # Also extract appendix (attachments) outside mainText
    m2 = re.search(r'<div class="appendix">(.*?)</div>', html, re.DOTALL)
    if m2:
        parts.append(m2.group(1).strip())
    if parts:
        return '<br><br>'.join(parts)
    return ''

def main():
    print(f'=== {SITE_NAME} ===')
    d = fetch_list(1, 1)
    total = d['elementsTotal']
    total_pages = (total + 14) // 15
    print(f'Total: {total}, Pages: {total_pages}')

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted, skipped, old = 0, 0, 0
    start = time.time()

    for pg in range(1, total_pages + 1):
        try:
            d = fetch_list(pg, 15)
        except Exception as e:
            print(f'  [ERROR] List page {pg}: {e}')
            time.sleep(2)
            continue
        arts = d.get('contents', [])
        if not arts:
            break

        for art in arts:
            title = (art.get('subject', '') or '').strip()
            pub_date = (art.get('fwdate', '') or '')[:10]
            if pub_date and pub_date < CUTOFF_DATE:
                old += 1
                continue
            xxid = art.get('xxid', '')
            dwid = art.get('dwid', '')
            dq = art.get('dq', '124')
            if not xxid or not title:
                skipped += 1
                continue

            try:
                content_html = fetch_detail(dq, dwid, xxid)
            except Exception as e:
                print(f'  [ERROR] Detail {xxid}: {e}')
                content_html = ''

            int_id = make_int_id(xxid)
            try:
                c.execute('INSERT OR IGNORE INTO gov_raw(id,site_name,source_url,page_url,title,publish_date,content,date_rank) VALUES(?,?,?,?,?,?,?,?)',
                          (int_id, SITE_NAME,
                           f'{BASE_URL}/{dq}/{dwid}/{xxid}.html',
                           f'{BASE_URL}/{dq}/{dwid}/{xxid}.html',
                           title, pub_date, content_html,
                           int(pub_date.replace('-','')) if pub_date else 0))
                if c.rowcount > 0: inserted += 1
                else: skipped += 1
            except Exception as e:
                print(f'  [DB ERR] {xxid}: {e}')
                skipped += 1

        elapsed = time.time() - start
        print(f'  Page {pg}/{total_pages}: +{inserted} (-{skipped}, >3y{old}) [{elapsed:.0f}s]')
        dates = [a.get('fwdate','')[:10] for a in arts]
        if dates and max(dates) < CUTOFF_DATE and pg > 1:
            print(f'  (Reached cutoff)')
            break

    conn.commit()
    print(f'\nSyncing FTS...')
    try:
        c2 = conn.cursor()
        c2.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
        rows = c2.execute("SELECT id, title, site_name, summary, content FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
        c2.executemany("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary, content) VALUES(?,?,?,?,?)", rows)
        conn.commit()
        print(f'  FTS: {len(rows)} records')
    except Exception as e:
        print(f'  FTS error: {e}')
    conn.close()
    elapsed = time.time() - start
    print(f'\n=== Done ({elapsed:.0f}s) === Inserted {inserted}, Skipped {skipped}, Old {old}')

if __name__ == '__main__':
    main()
