#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
五河县人民政府 - 建设项目环境影响评价审批 (wuhe.gov.cn/zfxxgk/public/column/29631?catId=18193258)
CMS: 安徽 zfxxgk public/column 平台 (label/8888 API, 同安庆sthjj)
列表: POST/GET /zfxxgk/site/label/8888  labelName=publicInfoList&siteId=6795621&organId=29631&catIds=18193258&type=4&pageIndex=N (15条/页)
详情: /zfxxgk/public/29631/{id}.html, 正文 .xxgkcontent / .gkwz_contnet
用法: python3 crawl_wuhe_zfxxgk.py [--pages=N] [--json out.jsonl]
"""
import re, sys, os, json, time, sqlite3, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = "五河县人民政府-建设项目环境影响评价审批"
BASE_URL = "https://www.wuhe.gov.cn"
API_URL = f"{BASE_URL}/zfxxgk/site/label/8888"
ORGAN_ID = "29631"
CAT_IDS = "18193258"
SITE_ID = "6795621"
DOMAIN = "www.wuhe.gov.cn"
GROUP = "安徽"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
           "Referer": f"{BASE_URL}/zfxxgk/public/column/{ORGAN_ID}?type=4&catId={CAT_IDS}&action=list&nav=3"}

def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)

def clean_title(t):
    t = re.sub(r"&nbsp;|&#160;|\u200b|\ufeff", "", t or "")
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"\.\.\.|…$", "", t).strip()
    return t

def fetch_list(page):
    params = {
        "labelName": "publicInfoList", "siteId": SITE_ID, "pageSize": "15",
        "pageIndex": str(page), "isDate": "true", "dateFormat": "yyyy-MM-dd",
        "length": "50", "active": "0", "organId": ORGAN_ID, "type": "4",
        "catIds": CAT_IDS, "fileNum": "", "filterFileNum": "", "fromCode": "",
        "sortType": "", "fuzzySearch": "", "keyWords": "", "orderType": "",
    }
    for attempt in range(4):
        try:
            r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            html = r.text
            if len(html) < 500:
                log(f"API P{page} small response ({len(html)}), retry")
                time.sleep(2 * (attempt + 1))
                continue
            items = []
            soup = BeautifulSoup(html, "html.parser")
            # 列表: 表格行 或 <a> 链接
            for a in soup.find_all("a", href=True):
                href = a.get("href", "")
                title = clean_title(a.get_text(strip=True))
                if not title or len(title) < 6:
                    continue
                if not re.search(r"/zfxxgk/public/\d+/\d+\.html", href):
                    continue
                full = href if href.startswith("http") else urljoin(BASE_URL, href)
                items.append({"title": title, "url": full})
            return items
        except Exception as e:
            if attempt == 3:
                log(f"API P{page} ERR: {e}")
                return []
            time.sleep(2 * (attempt + 1))
    return []

def extract_clean_text(content_html, page_url=None):
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")
    # 清干扰: 索引进度条/导航
    for tag in soup.find_all(string=re.compile(r"【字体：|字号：|打印")):
        p = tag.find_parent(["p", "div", "span", "td"])
        if p:
            p.decompose()
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True) or "附件"
        full_url = urljoin(page_url or BASE_URL, href)
        a_tag.replace_with(f'<p><a href="{full_url}">{text}</a></p>')
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()
    parts = []
    for el in soup.find_all(['table', 'p']):
        if el.name == 'table':
            parts.append(str(el))
        elif el.name == 'p' and not el.find_parent('table'):
            t = el.get_text(separator='', strip=True)
            if t:
                parts.append(f'<p>{t}</p>')
    return '\n\n'.join(parts) if parts else content_html.strip()

def parse_detail(url):
    for attempt in range(4):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            html = r.text
            break
        except Exception as e:
            if attempt == 3:
                log(f"detail ERR {url[-50:]}: {e}")
                return {"title": "", "publish_date": "", "content": ""}
            time.sleep(2 * (attempt + 1))
    soup = BeautifulSoup(html, "html.parser")
    result = {"title": "", "publish_date": "", "content": ""}
    ttag = soup.find("title")
    if ttag:
        result["title"] = clean_title(re.sub(r"_\w*$", "", ttag.get_text(strip=True)))
    m = re.search(r"(\d{4}-\d{2}-\d{2})", html[:8000])
    if m:
        result["publish_date"] = m.group(1)
    content_div = soup.find("div", class_="xxgkcontent")
    if not content_div:
        content_div = soup.find("div", class_="gkwz_contnet")
    if content_div:
        result["content"] = extract_clean_text(str(content_div), url)
    return result

def crawl(pages=5, json_out=None):
    log(f"Crawl pages={pages} cutoff={CUTOFF}")
    items_all = []
    seen = set()
    for p in range(1, pages + 1):
        items = fetch_list(p)
        new = 0
        for it in items:
            if it["url"] not in seen:
                seen.add(it["url"])
                items_all.append(it)
                new += 1
        log(f"P{p}: {len(items)} items ({new} new), total {len(items_all)}")
        if not items:
            break
        time.sleep(0.5)

    enriched = []
    for idx, it in enumerate(items_all, 1):
        d = parse_detail(it["url"])
        if d["publish_date"] and d["publish_date"] < CUTOFF:
            continue
        enriched.append({
            "title": d["title"] or it["title"],
            "page_url": it["url"],
            "publish_date": d["publish_date"],
            "content": d["content"],
            "site_name": SITE_NAME,
            "column": "建设项目环境影响评价审批",
        })
        if idx % 10 == 0:
            log(f"  {idx}/{len(items_all)}")
        time.sleep(0.3)

    if json_out:
        with open(json_out, "w", encoding="utf-8") as f:
            for item in enriched:
                f.write(json.dumps(item, ensure_ascii=False) + "\n")
        log(f"JSONL: {len(enriched)} items -> {json_out}")
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    stored = skipped = 0
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category, script_name)"
                " VALUES (?,?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], item["column"], "crawl_wuhe_zfxxgk.py"))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")

if __name__ == "__main__":
    args = sys.argv[1:]
    pages = 5
    json_out = None
    if "--pages=" in " ".join(args):
        pages = int(re.search(r"--pages=(\d+)", " ".join(args)).group(1))
    if "--json" in args:
        json_out = args[args.index("--json") + 1]
    crawl(pages=pages, json_out=json_out)
