#!/usr/bin/env python3
import os

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
"""淮南煤化工产业园-通知公告爬虫 ahccci.huainan.gov.cn"""
import re, sys, time, json, urllib.request, urllib.error, sqlite3, ssl
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

BASE_URL = "https://ahccci.huainan.gov.cn"
LIST_API = "https://ahccci.huainan.gov.cn/content/column/12494462"
SITE_NAME = "淮南煤化工产业园-通知公告"
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 3
DELAY = 1.0
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

ctx = ssl.create_default_context()
ctx.check_hostname = False
ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

def fetch(url, retries=3):
    for i in range(retries):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            with urllib.request.urlopen(req, context=ctx, timeout=20) as r:
                return r.read().decode("utf-8", errors="replace")
        except Exception as e:
            if i < retries - 1:
                time.sleep(DELAY * (i+1))
                continue
            return None

def parse_list(html):
    """Extract (url, title, date) from list HTML"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_=re.compile(r"doc_list"))
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a")
        date_span = li.find("span", class_="date")
        if a and a.get("href") and date_span:
            url = a["href"].strip()
            if not url.startswith("http"):
                url = BASE_URL + url
            title = a.get("title", a.get_text(strip=True))
            date = date_span.get_text(strip=True)
            if re.match(r"\d{4}-\d{2}-\d{2}", date):
                items.append((url, title, date))
    return items

def parse_detail(html, url):
    """Extract content from detail page"""
    soup = BeautifulSoup(html, "html.parser")
    
    title_el = soup.select_one("h1.wztit")
    title = title_el.get_text(strip=True) if title_el else ""
    
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    date = meta_date["content"][:10] if meta_date and meta_date.get("content") else ""
    
    meta_source = soup.find("meta", attrs={"name": "ContentSource"})
    source = meta_source["content"] if meta_source and meta_source.get("content") else ""
    
    content_div = soup.select_one("#zoom.wzcon, div.wzcon")
    content = ""
    if content_div:
        # Remove unnecessary elements
        for el in content_div.find_all(["script", "style"]):
            el.decompose()
        content = str(content_div)
    
    return title, date, source, content

def save_to_db(items):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    inserted = 0
    for url, title, date, source, content in items:
        try:
            summary = f"来源：{source}" if source else ""
            c.execute("""
                INSERT OR IGNORE INTO gov_raw 
                (page_url, title, publish_date, site_name, source_url, summary, content)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (url, title, date, SITE_NAME, url, summary, content))
            if c.rowcount > 0:
                inserted += 1
            else:
                c.execute("UPDATE gov_raw SET content=? WHERE page_url=? AND (content IS NULL OR content='')",
                         (content, url))
                if c.rowcount > 0:
                    inserted += 1
        except Exception as e:
            print(f"  [ERROR] {url[:60]} - {e}", flush=True)
    conn.commit()
    conn.close()
    return inserted

def main():
    print(f"[hn] 淮南煤化工产业园-通知公告 - 截止: {CUTOFF_DATE}", flush=True)
    
    all_items = []
    for page in range(min(MAX_PAGES, _MAX_PG or MAX_PAGES)):
        if page == 0:
            url = f"{BASE_URL}/xwzx/tzgg/index.html"
        else:
            url = f"{LIST_API}?pageIndex={page}"
        
        print(f"[hn] 列表 {page+1}: {url}", flush=True)
        html = fetch(url)
        if not html:
            print(f"  [SKIP] 获取失败", flush=True)
            continue
        
        items = parse_list(html)
        if not items:
            print(f"  [END] 空列表", flush=True)
            break
        
        dates = [d for _,_,d in items]
        oldest = min(dates)
        newest = max(dates)
        print(f"  → {len(items)}条, {oldest} ~ {newest}", flush=True)
        
        for url, title, date in items:
            if date >= CUTOFF_DATE:
                all_items.append((url, title, date))
        
        if oldest < CUTOFF_DATE:
            print(f"  [END] {oldest} < 截止", flush=True)
            break
        time.sleep(DELAY)
    
    print(f"\n[hn] 共收集 {len(all_items)} 条", flush=True)
    
    # 预去重: 已有正文的 URL 不再抓详情 (避免每天重抓全量 → 超时)
    try:
        _c = sqlite3.connect(DB_PATH, timeout=60)
        _have = set(r[0] for r in _c.execute(
            "SELECT page_url FROM gov_raw WHERE site_name=? AND COALESCE(content,'')<>''",
            (SITE_NAME,)))
        _c.close()
        _before = len(all_items)
        all_items = [it for it in all_items if it[0] not in _have]
        print(f"[hn] 预去重: {_before} → {len(all_items)} (库中已有正文 {len(_have)})", flush=True)
    except Exception as e:
        print(f"[hn] 预去重跳过: {str(e)[:70]}", flush=True)

    detailed = []
    for i, (url, title, date) in enumerate(all_items, 1):
        print(f"[hn] 详情 {i}/{len(all_items)}: {title[:30]}...", flush=True)
        html = fetch(url)
        if not html:
            detailed.append((url, title, date, "", ""))
            continue
        dt, dd, ds, dc = parse_detail(html, url)
        ft = dt or title
        fd = dd or date
        cl = len(dc) if dc else 0
        print(f"  → 正文 {cl}字", flush=True)
        detailed.append((url, ft, fd, ds, dc))
        time.sleep(DELAY)
    
    inserted = save_to_db(detailed)
    print(f"\n[hn] ✅ 入库 {inserted} 条", flush=True)
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    try:
        conn.execute("INSERT INTO gov_search(gov_search, rowid, title, site_name, summary) VALUES('rebuild', 0, '', '', '')")
    except:
        pass
    conn.commit()
    conn.close()
    print(f"[hn] ✅ 完成", flush=True)

if __name__ == "__main__":
    main()
