#!/usr/bin/env python3
"""crawl_ybs.py — 赤峰市元宝山区人民政府 通知公告 (TRS WCM)"""
import os, re, sys, time, requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
import sqlite3

SEARCH_DB = os.getenv("SEARCH_DB", "/mnt/data/search.db")
SITE_NAME = "元宝山区-通知公告"
CATEGORY = "通知公告"
BASE_URL = "http://www.ybs.gov.cn/bsdt/tzgg/"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

is_incremental = len(sys.argv) >= 2
if is_incremental:
    cutoff_days = int(sys.argv[1])
    cutoff = (datetime.now() - timedelta(days=cutoff_days)).strftime("%Y-%m-%d")
else:
    cutoff = CUTOFF_DATE

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print("  [ERROR] fetch %s: %s" % (url, e))
        return ""

def safe_summary(text, max_len=500):
    if len(text) <= max_len:
        return text
    truncated = text[:max_len]
    if '<' in truncated:
        last_open = truncated.rfind('<')
        last_close = truncated.rfind('>')
        if last_open > last_close:
            truncated = truncated[:last_open]
    return truncated

def extract_content(html):
    """Extract content from detail page."""
    m = re.search(r'<div class="deatil-detail">(.*?)</div>\s*</div>\s*</div>',
                  html, re.DOTALL)
    if m:
        txt = m.group(1)
    else:
        m = re.search(r'<div class="trs_editor_view[^"]*".*?</style>(.*?)</div>',
                      html, re.DOTALL)
        if m:
            txt = m.group(1)
        else:
            return ""
    txt = re.sub(r'<script[^>]*>.*?</script>', '', txt, flags=re.DOTALL)
    txt = re.sub(r'<style[^>]*>.*?</style>', '', txt, flags=re.DOTALL)
    txt = re.sub(r'<br\s*/?>', '\n', txt)
    txt = re.sub(r'</p>', '\n\n', txt)
    txt = re.sub(r'</div>', '\n\n', txt)
    table_placeholders = []
    def save_table(m):
        ph = '__TABLE_%d__' % len(table_placeholders)
        table_placeholders.append((ph, m.group(0)))
        return ph
    txt = re.sub(r'<table[^>]*>.*?</table>', save_table, txt, flags=re.DOTALL)
    txt = re.sub(r'<[^>]+>', '', txt)
    txt = re.sub(r'\n{3,}', '\n\n', txt)
    for ph, ht in table_placeholders:
        txt = txt.replace(ph, ht)
    return txt.strip()

def get_list_items(html):
    items = []
    soup = BeautifulSoup(html, 'lxml')
    ul = soup.find('ul', id='lb')
    if not ul:
        return items
    for li in ul.find_all('li', class_='clearfix'):
        a = li.find('a')
        span = li.find('span')
        if a and span:
            href = a.get('href', '')
            if href.startswith('./'):
                href = BASE_URL + href[2:]
            elif not href.startswith('http'):
                href = BASE_URL + href.lstrip('/')
            title = a.get_text(strip=True)
            date = span.get_text(strip=True)
            if title and date:
                items.append((href, title, date))
    return items

def get_total_pages(html):
    m = re.search(r'var countPage\s*=\s*(\d+);', html)
    if m:
        return int(m.group(1))
    return 1

def main():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    total_new = 0
    total_skip = 0

    html = fetch(BASE_URL + "index.html")
    if not html:
        print("[ERROR] Cannot fetch page 1")
        return
    total_pages = get_total_pages(html)
    mode = "增量" if is_incremental else "全量"
    print("[%s] 共 %d 页 (%s, cutoff=%s)" % (SITE_NAME, total_pages, mode, cutoff))

    for page in range(total_pages):
        if page == 0:
            url = BASE_URL + "index.html"
        else:
            url = BASE_URL + "index_%d.html" % page

        if page > 0:
            html = fetch(url)
            if not html:
                print("  [SKIP] 第%d页 无法获取" % (page+1))
                continue

        items = get_list_items(html)
        if not items:
            print("  [EMPTY] 第%d页 无数据" % (page+1))
            continue

        page_new = 0
        for href, title, date_str in items:
            if date_str < cutoff:
                continue

            c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,))
            if c.fetchone():
                total_skip += 1
                continue

            detail_html = fetch(href)
            if not detail_html:
                total_skip += 1
                continue

            content = extract_content(detail_html)
            if not content or len(content) < 20:
                total_skip += 1
                continue

            summary = safe_summary(content)

            c.execute("""INSERT INTO gov_raw (site_name, category, title, page_url, content, summary, publish_date)
                         VALUES (?, ?, ?, ?, ?, ?, ?)""",
                      (SITE_NAME, CATEGORY, title, href, content, summary, date_str))
            page_new += 1
            total_new += 1

        conn.commit()
        print("  第%d/%d页: +%d (累计%d)" % (page+1, total_pages, page_new, total_new))

    conn.close()
    print("\n[%s] 完成: 新增 %d, 跳过 %d" % (SITE_NAME, total_new, total_skip))

if __name__ == "__main__":
    main()
