#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""crawl_lixin_jcsd.py - 利辛县-基层政务公开专题 (www.lixin.gov.cn)
站点: 静态HTML, 无WAF
列表: /Jcsd/showList/{subId}/page_{N}.html (21个环保子栏目, 专题首页 111000000)
  子栏目: 111001001环评审批 111001002固废许可 111001003危废许可 111001004排污许可
          111001005夜间施工 111001006排污口审批 111002000行政处罚 111003000行政强制
          111004000行政规划 111006000环境信访 111007000其他权力 111007001现场检查
          111007002清洁生产 111007003专项资金 111008000公共服务 111008001环保宣传
          111008002环境质量 111008003重污染天气 111008004固废危废 111009000污染综合防治
详情: /Jcsd/show/{id}.html → div#zoom.j-fontContent + meta ArticleTitle/PubDate
运行: python3 crawl_lixin_jcsd.py [--pages=N] [--sub=111001001,...] [--dryrun]
"""
import sys, os, re, json, time, requests, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

SITE_NAME = "利辛县-基层政务公开"
BASE = "https://www.lixin.gov.cn"
MAX_PAGES_DEFAULT = 5
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
OUT = "/tmp/lixin_jcsd.jsonl"
DB_PATH = "/root/search.db"
SCRIPT_NAME = "crawl_lixin_jcsd.py"

SUB_IDS = [
    "111001001", "111001002", "111001003", "111001004", "111001005", "111001006",
    "111002000", "111003000", "111004000", "111006000", "111007000", "111007001",
    "111007002", "111007003", "111008000", "111008001", "111008002", "111008003",
    "111008004", "111009000",
]


def parse_pages_arg():
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            return int(a.split("=")[1])
    return MAX_PAGES_DEFAULT


def get_sub_filter():
    for a in sys.argv[1:]:
        if a.startswith("--sub="):
            return a.split("=")[1].split(",")
    return None


def clean_title(t):
    return re.sub(r"^[•·\s]+", "", t or "").strip().replace("\u200b", "").replace("\ufeff", "")


def fetch(url, retries=2):
    for i in range(retries):
        try:
            r = requests.get(url, headers={"User-Agent": UA}, timeout=20, verify=False)
            r.encoding = "utf-8"
            return r.text
        except Exception:
            time.sleep(2)
    return ""


def list_items(sub_id, page):
    """返回 [(title, url, date)]"""
    url = f"{BASE}/Jcsd/showList/{sub_id}/page_{page}.html"
    html = fetch(url)
    if not html:
        return []
    items = []
    # 文章链接: /Jcsd/show/{id}.html
    for m in re.finditer(r'href="(/Jcsd/show/(\d+)\.html)"[^>]*>\s*([^<]{8,100})', html):
        href, id_, title = m.group(1), m.group(2), clean_title(m.group(3))
        items.append({"title": title, "url": BASE + href})
    return items


def extract_block(el):
    """递归提取 HTML 块 (保持顺序): p/pre 保留内部HTML, table 保留, div 递归"""
    if el.name == "table":
        return [str(el)]
    if el.name in ("p", "pre"):
        if el.find_parent("table"):
            return []
        txt = el.get_text(" ", strip=True)
        if not txt or re.match(r"^(责任编辑|初审|复审|终审|\[纠错\]|浏览次数|字号|分享)", txt):
            return []
        return [re.sub(r"\s+", " ", str(el)).strip()]
    if el.name == "div":
        blocks = []
        for c in el.find_all(["p", "pre", "table", "div"], recursive=False):
            blocks.extend(extract_block(c))
        if not blocks:
            txt = el.get_text(" ", strip=True)
            if txt:
                blocks.append(f"<p>{txt}</p>")
        return blocks
    return []


def parse_detail(url):
    html = fetch(url)
    if not html:
        return "", "", ""
    title = ""
    m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    if m:
        title = m.group(1).strip()
    if not title:
        m = re.search(r"<title>([^<]*)</title>", html, re.S)
        if m:
            title = re.sub(r"\s*_?利辛县人民政府.*$", "", m.group(1)).strip()
    date_text = ""
    m = re.search(r'<meta name="PubDate" content="(\d{4}-\d{2}-\d{2})', html)
    if m:
        date_text = m.group(1)
    content = ""
    soup = BeautifulSoup(html, "html.parser")
    zoom = soup.find("div", id="zoom")
    parts = []
    if zoom:
        for el in zoom.find_all(["p", "pre", "table", "div"], recursive=False):
            parts.extend(extract_block(el))
        dedup = []
        for p in parts:
            if p and (not dedup or dedup[-1] != p):
                dedup.append(p)
        content = "\n\n".join(dedup)
    # 附件: zoom 内带扩展名 + 全页 m-dtdownload 下载区 (href 可能无扩展名)
    atts = []
    for a in soup.find_all("a", href=True):
        h = a.get("href", "")
        if re.search(r"\.(docx?|pdf|xlsx?|rar|zip|wps|et|dps)(\?|$)", h, re.I) or "/upload_bz/download" in h:
            name = re.sub(r"【.*?】", "", a.get_text(strip=True)).strip() or h.split("/")[-1]
            if h.startswith("/"):
                h = BASE + h
            elif not h.startswith("http"):
                h = BASE + "/" + h
            atts.append({"name": name, "url": h})
    for a in atts:
        link = f'<p><a href="{a["url"]}" target="_blank">{a["name"]}</a></p>'
        if link not in content:
            content += "\n" + link
    return title, date_text, content


def main():
    max_pages = parse_pages_arg()
    sub_filter = get_sub_filter()
    dryrun = "--dryrun" in sys.argv
    print(f"[lixin_jcsd] pages={max_pages} subs={len(sub_filter) if sub_filter else len(SUB_IDS)} cutoff={CUTOFF}", flush=True)
    results = []
    subs = sub_filter if sub_filter else SUB_IDS
    for sub_id in subs:
        for pn in range(1, max_pages + 1):
            items = list_items(sub_id, pn)
            if not items:
                break
            print(f"  [sub={sub_id}] page{pn}: {len(items)} 条", flush=True)
            for it in items:
                try:
                    title, date_text, content = parse_detail(it["url"])
                    if date_text and date_text < CUTOFF:
                        continue
                    results.append({
                        "url": it["url"], "title": title or it["title"],
                        "date": date_text, "content": content,
                        "sub": sub_id,
                    })
                    print(f"    ✓ {results[-1]['title'][:40]} | {date_text}", flush=True)
                except Exception as e:
                    print(f"    ✗ {str(e)[:60]}", flush=True)
                time.sleep(0.5)
    print(f"[完成] {len(results)} 条", flush=True)
    if dryrun:
        for r in results[:5]:
            print(json.dumps(r, ensure_ascii=False)[:200])
        return
    with open(OUT, "w", encoding="utf-8") as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + "\n")
    if os.path.exists(DB_PATH):
        conn = sqlite3.connect(DB_PATH, timeout=120)
        conn.execute("PRAGMA busy_timeout=120000")
        c = conn.cursor()
        added = 0
        for r in results:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (r["url"],))
            if c.fetchone():
                continue
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank, script_name) VALUES (?,?,?,?,?,?,?,?,?,?)",
                (r["title"], SITE_NAME, "安徽", r["url"], r["date"], r["content"], "", "",
                 int(r["date"].replace("-", "")) if re.match(r"\d{4}-\d{2}-\d{2}", r["date"]) else 0, SCRIPT_NAME))
            if c.rowcount:
                added += 1
        conn.commit()
        conn.close()
        print(f"[入库] 新增 {added}", flush=True)


if __name__ == "__main__":
    main()
