#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
本溪满族自治县 - 公示公告 (bx.gov.cn/zwgk/zfgg)
CMS: TRS WCM (content_666492)
列表: a[title] 含"标题：xxx 点击数：N 发表时间：date"
分页: /zwgk/zfgg_2, /zwgk/zfgg_3 (无参数直接可访问)
详情: /zwgk/zfgg/content_666492, 容器 id=content / class=conTxt / printArea
用法: python3 crawl_bx_gggs.py [--pages=N] [--json out.jsonl]
"""
import re, sys, os, json, time, sqlite3, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = "本溪满族自治县-公示公告"
BASE_URL = "http://www.bx.gov.cn"
LIST_URL = "http://www.bx.gov.cn/zwgk/zfgg"
DOMAIN = "www.bx.gov.cn"
GROUP = "辽宁"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"}


def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)


def clean_title(t):
    t = re.sub(r"&nbsp;|&#160;|\u200b|\ufeff", "", t or "")
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"\.\.\.|…$", "", t).strip()
    return t


def fetch(url, timeout=30):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = "utf-8"
    return r.text


def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for a in soup.find_all("a", href=True):
        href = a.get("href", "")
        if "/content_" not in href:
            continue
        title = clean_title(a.get("title", ""))
        if not title:
            continue
        # title attr like: 标题：xxx\n点击数：3\n发表时间：2026-08-13
        pub_date = ""
        m = re.search(r"发表时间[：:]\s*(20\d{2}[-/年]\d{1,2}[-/月]\d{1,2})", title)
        if m:
            pub_date = m.group(1).replace("年", "-").replace("月", "-").replace("/", "-")[:10]
        title = re.sub(r"^标题[：:]\s*", "", title).strip()
        # 去尾部 "点击数：N 发表时间：date"
        title = re.sub(r"\s*点击数[：:]\s*\d+\s*发表时间[：:]\s*20\d{2}[-/年]\d{1,2}[-/月]\d{1,2}\s*$", "", title).strip()
        items.append({"title": title, "url": urljoin(BASE_URL, href), "pub_date": pub_date})
    seen = set()
    uniq = []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            uniq.append(it)
    return uniq


def parse_detail(html):
    result = {"title": "", "publish_date": "", "content": ""}
    soup = BeautifulSoup(html, "html.parser")
    # h1/h2 优先（遍历取第一个有效完整标题），title 标签兜底
    for h in soup.find_all(["h1", "h2"]):
        t = h.get_text(strip=True)
        if len(t) >= 6 and not t.startswith(("您当前的位置", "当前位置", "首页", "网站")) and "点击数" not in t:
            result["title"] = clean_title(t)
            break
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        result["publish_date"] = meta_date["content"][:10]
    if not result["publish_date"]:
        m = re.search(r"20\d{2}[-/年]\d{1,2}[-/月]\d{1,2}", html[:8000])
        if m:
            result["publish_date"] = m.group(0).replace("年", "-").replace("月", "-").replace("/", "-")[:10]
    # content: id=content or class=conTxt or printArea
    content_html = ""
    for cid in ["content", "conTxt"]:
        div = soup.find("div", id=cid) or soup.find("div", class_=cid)
        if div:
            content_html = str(div)
            break
    if not content_html:
        pa = soup.find("div", class_="printArea")
        if pa:
            content_html = str(pa)
    if content_html:
        result["content"] = extract_clean_text(content_html)
    return result


def extract_clean_text(content_html):
    """输出纯 HTML：段落 <p>、链接 <a href target=_blank>、表格保留、图片 <img>、附件独立成段。"""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")
    for tag in soup.find_all(['script', 'style']):
        tag.decompose()
    # PDF 预览图 (data-uitype="pdf" 或 data-powerurl 指向 .pdf) → 转附件链接，删除预览图
    pdf_links = []
    for img in soup.find_all('img'):
        purl = img.get('data-powerurl', '') or img.get('data-org', '')
        if img.get('data-uitype') == 'pdf' or purl.lower().endswith('.pdf'):
            href = purl or img.get('src', '')
            if href and not href.startswith(('javascript:', '#')):
                pdf_links.append((href, urljoin(BASE_URL, href)))
            img.decompose()
    # 过滤装饰图标
    for img in soup.find_all('img'):
        src0 = img.get('src', '')
        if any(k in src0 for k in ('un-collect', 'collect', 'share', 'favicon', 'qrcode', 'ewm', 'banner', 'top_')):
            img.decompose()
    # 附件区 ul/li 扁平化
    attach_links = []
    for li in soup.find_all('li'):
        a_in = li.find('a', href=True)
        if a_in and not li.find_parent('table'):
            href = a_in.get('href', '').strip()
            text = a_in.get_text(strip=True) or os.path.basename(href.split('?')[0]) or '附件'
            if href and not href.startswith(('javascript:', '#')):
                attach_links.append((text, urljoin(BASE_URL, href)))
    for li in soup.find_all('li'):
        if li.find('a', href=True) and not li.find_parent('table'):
            li.decompose()
    # 剥装饰标签
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()
    # 移除纯装饰空标签
    for tag in soup.find_all(['div', 'p']):
        if not tag.get_text(strip=True) and not tag.find(['img', 'table', 'a']):
            tag.decompose()
    from html import escape
    parts = []
    for el in soup.find_all(['table', 'p']):
        if el.name == 'table':
            parts.append(str(el))
        elif el.name == 'p' and not el.find_parent('table'):
            inner = []
            for c in el.contents:
                name = getattr(c, "name", None)
                if name == 'img':
                    src0 = c.get('src', '')
                    if any(k in src0.lower() for k in ('.gif', 'icon16', 'banner')):
                        continue
                    inner.append(str(c))
                elif name in ('a', 'table', 'br'):
                    inner.append(str(c))
                elif name:
                    inner.append(str(c))
                else:
                    inner.append(escape(str(c)))
            t = "".join(inner).strip()
            if t:
                parts.append(f"<p>{t}</p>")
    # 图片独立成段（非装饰）
    for img in soup.find_all('img'):
        src0 = img.get('src', '')
        if img.find_parent('p') or img.find_parent('table'):
            continue
        if any(k in src0.lower() for k in ('.gif', 'icon16', 'banner')):
            continue
        src = src0.strip()
        if src:
            parts.append('<p><img src="%s"></p>' % escape(urljoin(BASE_URL, src)))
    seen_a = set()
    for text, url in attach_links:
        key = (text, url)
        if key in seen_a:
            continue
        seen_a.add(key)
        parts.append(f'<p><a href="{url}" target="_blank">{text}</a></p>')
    # PDF 预览图转的附件链接（去重）
    seen_pdf = set()
    for href, url in pdf_links:
        if url in seen_pdf:
            continue
        seen_pdf.add(url)
        fname = os.path.basename(url.split('?')[0]) or '附件'
        parts.append(f'<p><a href="{url}" target="_blank">{fname}</a></p>')
    return "\n\n".join(parts) if parts else escape(content_html).strip()


def crawl(pages=1, json_out=None):
    log(f"Crawl pages={pages} cutoff={CUTOFF}")
    items_all = []
    seen = set()
    for p in range(1, pages + 1):
        url = LIST_URL if p == 1 else f"{LIST_URL}_{p}"
        try:
            html = fetch(url)
        except Exception as e:
            log(f"P{p} ERR: {e}")
            break
        items = parse_list(html)
        new = 0
        for it in items:
            if it["url"] not in seen:
                seen.add(it["url"])
                items_all.append(it)
                new += 1
        log(f"P{p}: {len(items)} items ({new} new), total {len(items_all)}")
        if not items:
            break
        time.sleep(0.8)

    enriched = []
    for idx, it in enumerate(items_all, 1):
        if it["pub_date"] and it["pub_date"] < CUTOFF:
            continue
        try:
            html = fetch(it["url"])
            detail = parse_detail(html)
        except Exception as e:
            log(f"  detail {it['url'][-40:]} ERR: {e}")
            detail = {"title": "", "publish_date": "", "content": ""}
        enriched.append({
            "title": detail["title"] or it["title"],
            "page_url": it["url"],
            "publish_date": detail["publish_date"] or it["pub_date"],
            "content": detail["content"],
            "site_name": SITE_NAME,
            "column": "公示公告",
        })
        if idx % 10 == 0:
            log(f"  {idx}/{len(items_all)}")
        time.sleep(0.4)

    if json_out:
        with open(json_out, "w", encoding="utf-8") as f:
            for item in enriched:
                f.write(json.dumps(item, ensure_ascii=False) + "\n")
        log(f"JSONL: {len(enriched)} items -> {json_out}")
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    stored = skipped = 0
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], item["column"]))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


if __name__ == "__main__":
    args = sys.argv[1:]
    pages = 5
    json_out = None
    if "--pages=" in " ".join(args):
        pages = int(re.search(r"--pages=(\d+)", " ".join(args)).group(1))
    if "--json" in args:
        json_out = args[args.index("--json") + 1]
    crawl(pages=pages, json_out=json_out)
