#!/usr/bin/env python3
"""Crawl 曲靖市麒麟区 - 公示公告
http://www.ql.gov.cn/gov/public/info/tzgg.html
"""
import requests, re, sqlite3, sys, time, hashlib
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import os

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '麒麟区 - 公示公告'
BASE_URL = 'http://www.ql.gov.cn'
LIST_URL = 'http://www.ql.gov.cn/gov/public/info/tzgg.html'
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
MAX_PAGES = 5

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}
session = requests.Session()
session.headers.update(HEADERS)

def init_session():
    """Get past WAF by visiting homepage first"""
    try:
        resp = session.get(BASE_URL, timeout=30)
        return resp.status_code == 200
    except:
        return False

def parse_list(html):
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='p-list')
    if not ul:
        return items
    for li in ul.find_all('li', recursive=False):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href']
        if not href.startswith('/gov/public/detail/'):
            continue
        title = a.get('title') or a.get_text(strip=True)
        # Date is text after </a> in the li
        li_text = li.get_text()
        a_text = a.get_text()
        date = li_text.replace(a_text, '', 1).strip()
        dm = re.search(r'(\d{4}-\d{2}-\d{2})', date)
        date_str = dm.group(1) if dm else ''
        url = BASE_URL + href
        if title and date_str:
            items.append({'title': title.strip(), 'url': url, 'date': date_str})
    return items

def fetch_detail(url):
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except:
        return '', '', ''

    soup = BeautifulSoup(html, 'html.parser')

    # Title from h3.public_web_title
    h3 = soup.find('h3', class_='public_web_title')
    title = h3.get_text(strip=True) if h3 else ''

    # Date from "发布时间" text
    date_str = ''
    dm = re.search(r'发布时间\s*(\d{4}-\d{2}-\d{2})', html)
    if dm:
        date_str = dm.group(1)

    # Content from div.public_web_con.des
    con = soup.find('div', class_='public_web_con')
    if con:
        content_raw = str(con)
        content_raw = re.sub(r'<script[^>]*>.*?</script>', '', content_raw, flags=re.DOTALL|re.I)
        content_raw = re.sub(r'<style[^>]*>.*?</style>', '', content_raw, flags=re.DOTALL|re.I)
        return title, date_str, content_raw.strip()

    return title, date_str, ''

def main():
    is_incremental = 'incremental' in sys.argv
    print(f'=== {SITE_NAME} ===')

    if not init_session():
        print('  [ERROR] Failed to init session (WAF?)')
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted, skipped, old_total = 0, 0, 0
    max_pages = 1 if is_incremental else MAX_PAGES
    consecutive_old = 0
    min_pages = 5
    start_time = time.time()

    for pg in range(1, max_pages + 1):
        if pg == 1:
            list_url = LIST_URL
        else:
            list_url = f'{LIST_URL}?cname=tzgg&page={pg}'
        try:
            resp = session.get(list_url, timeout=30)
            resp.encoding = 'utf-8'
            raw_html = resp.text
        except Exception as e:
            print(f'  [ERROR] Page {pg}: {e}')
            time.sleep(2)
            continue

        items = parse_list(raw_html)
        if not items:
            print(f'  Page {pg}: 0 items (end)')
            break

        all_old = all(item['date'] < CUTOFF_DATE for item in items if item['date'])
        if all_old:
            consecutive_old += 1
            if consecutive_old >= 2 and pg >= min_pages:
                print(f'  Page {pg}: all old, stopping')
                break
        else:
            consecutive_old = 0

        for item in items:
            if item['date'] and item['date'] < CUTOFF_DATE:
                old_total += 1
                continue

            has = c.execute("SELECT 1 FROM gov_raw WHERE page_url=? AND content IS NOT NULL AND content!=''", (item['url'],)).fetchone()
            if has:
                skipped += 1
                continue

            try:
                title, date_str, content = fetch_detail(item['url'])
            except Exception as e:
                print(f'  [ERROR] Detail: {e}')
                time.sleep(1)
                skipped += 1
                continue

            final_title = title or item['title']
            final_date = date_str or item['date']
            int_id = int(hashlib.md5(item['url'].encode()).hexdigest()[:15], 16) % (2**63)
            date_rank = int(final_date.replace('-', '')) if final_date else 0
            summary = re.sub(r'<[^>]+>', '', content)[:200] if content else final_title
            summary = re.sub(r'\s+', ' ', summary).strip()

            try:
                c.execute(
                    'INSERT OR IGNORE INTO gov_raw(id, site_name, source_url, page_url, title, publish_date, content, summary, date_rank) VALUES(?,?,?,?,?,?,?,?,?)',
                    (int_id, SITE_NAME, item['url'], item['url'], final_title, final_date, content, summary, date_rank)
                )
                if c.rowcount > 0:
                    inserted += 1
            except Exception as e:
                print(f'  [DB] {e}')
                skipped += 1
            time.sleep(0.3)

        elapsed = time.time() - start_time
        print(f'  Page {pg}/{max_pages}: +{inserted} (skipped {skipped}, >3y {old_total}) [{elapsed:.0f}s]')

    conn.commit()
    # FTS rebuild
    try:
        c2 = conn.cursor()
        # QC20260926 去掉手写 gov_search 整站删除(抢锁源; FTS 由 gov_raw 触发器维护) 
        # c2.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name=?)", (SITE_NAME,))
        rows = c2.execute("SELECT id, title, site_name, summary FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
        for r in rows:
            c2.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES(?,?,?,?)", r)
        conn.commit()
        print(f'  FTS: {len(rows)} records rebuilt')
    except Exception as e:
        print(f'  FTS error: {e}')
    conn.close()
    elapsed = time.time() - start_time
    print(f'\nDone ({elapsed:.0f}s). Inserted {inserted}, Skipped {skipped}, Old {old_total}')

if __name__ == '__main__':
    main()
