#!/usr/bin/env python3
"""Crawl 鄂托克前旗人民政府 - 通知公告
http://www.etkqq.gov.cn/xwdt/xwdt_tzgg/
"""
import requests, re, sqlite3, sys, time, hashlib
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '鄂托克前旗 - 通知公告'
BASE_URL = 'http://www.etkqq.gov.cn'
LIST_PATH = '/xwdt/xwdt_tzgg/'
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
MAX_PAGES = 5

HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
session = requests.Session()
session.headers.update(HEADERS)

def parse_list(html):
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    for ul in soup.find_all('ul'):
        lis = ul.find_all('li', recursive=False)
        if len(lis) < 5:
            continue
        articles = []
        for li in lis:
            a = li.find('a', href=True)
            if not a:
                continue
            href = a['href']
            if href.startswith('http') and 'etkqq' not in href:
                continue
            raw = a.get_text(strip=True)
            if not raw or len(raw) < 10:
                continue
            # Extract date from end of title: "XXXX...2026-06-11"
            dm = re.search(r'(\d{4}-\d{2}-\d{2})\s*$', raw)
            date_str = dm.group(1) if dm else ''
            title = raw[:dm.start()].strip() if dm else raw
            url = href if href.startswith('http') else BASE_URL + LIST_PATH.rstrip('/') + '/' + href.lstrip('./')
            if title:
                articles.append({'title': title, 'url': url, 'date': date_str})
        if len(articles) >= 3:
            return articles
    return items

def fetch_detail(url):
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except:
        return '', '', ''
    tm = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    title = tm.group(1).strip() if tm else ''
    dm = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
    date_str = dm.group(1)[:10] if dm else ''
    soup_detail = BeautifulSoup(html, 'html.parser')
    trs = soup_detail.find('div', class_='TRS_Editor')
    if trs:
        for tag in trs.find_all(['script', 'style']):
            tag.decompose()
        content = str(trs).strip()
        return title, date_str, content
    return title, date_str, ''

def get_page_url(pg):
    if pg == 1:
        return BASE_URL + LIST_PATH
    return f'{BASE_URL}{LIST_PATH}index_{pg-1}.html'

def main():
    is_incremental = 'incremental' in sys.argv
    print(f'=== {SITE_NAME} ===')
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    inserted = skipped = old_total = 0
    max_pages = 1 if is_incremental else MAX_PAGES
    consecutive_old = 0
    start = time.time()
    for pg in range(1, max_pages + 1):
        try:
            resp = session.get(get_page_url(pg), timeout=30)
            resp.encoding = 'utf-8'
            html = resp.text
        except Exception as e:
            print(f'  [ERR] Page {pg}: {e}')
            time.sleep(2)
            continue
        items = parse_list(html)
        if not items:
            print(f'  Page {pg}: 0 items (end)')
            break
        all_old = all(it['date'] and it['date'] < CUTOFF_DATE for it in items if it['date'])
        if all_old:
            consecutive_old += 1
            if consecutive_old >= 2 and pg >= 5:
                print(f'  Page {pg}: all old, stop')
                break
        else:
            consecutive_old = 0
        for it in items:
            if it['date'] and it['date'] < CUTOFF_DATE:
                old_total += 1
                continue
            has = c.execute("SELECT 1 FROM gov_raw WHERE page_url=? AND content IS NOT NULL AND content!=''", (it['url'],)).fetchone()
            if has:
                skipped += 1
                continue
            try:
                title, date_str, content = fetch_detail(it['url'])
            except Exception as e:
                print(f'  [ERR] Detail: {e}')
                time.sleep(1)
                skipped += 1
                continue
            final_title = title or it['title']
            final_date = date_str or it['date']
            fid = int(hashlib.md5(it['url'].encode()).hexdigest()[:15], 16) % (2**63)
            dr = int(final_date.replace('-','')) if final_date else 0
            summary = re.sub(r'<[^>]+>', '', content)[:200] if content else final_title
            summary = re.sub(r'\s+', ' ', summary).strip()
            c.execute('INSERT OR IGNORE INTO gov_raw(id,site_name,source_url,page_url,title,publish_date,content,summary,date_rank) VALUES(?,?,?,?,?,?,?,?,?)',
                      (fid, SITE_NAME, it['url'], it['url'], final_title, final_date, content, summary, dr))
            if c.rowcount > 0:
                inserted += 1
            time.sleep(0.3)
        print(f'  Page {pg}/{max_pages}: +{inserted} (skip {skipped}, old {old_total}) [{time.time()-start:.0f}s]')
    conn.commit()
    c2 = conn.cursor()
    c2.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
    rows = c2.execute("SELECT id,title,site_name,summary FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
    for r in rows:
        c2.execute("INSERT OR IGNORE INTO gov_search(rowid,title,site_name,summary) VALUES(?,?,?,?)", r)
    conn.commit()
    conn.close()
    print(f'\nDone ({time.time()-start:.0f}s). +{inserted}, skip {skipped}, old {old_total}')

if __name__ == '__main__':
    main()
