#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
广德市人民政府 - 通知公告 爬虫
栏目: https://www.guangde.gov.cn/News/showList/933/page_1.html
CMS: 自定义PHP政府信息公开平台
列表: /News/showList/933/page_{n}.html (15条/页)
详情: /News/show/{id}.html
  标题: <meta name="ArticleTitle"> (完整)
  正文: <div id="zoom" class="detail_content clearfix"> 保留 <p>/<table> HTML
  日期: <meta name="PubDate">
用法:
  python3 crawl_guangde_notice.py --pages=5
  python3 crawl_guangde_notice.py --pages=1
"""
import os, sys, re, time, json, html as html_mod
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE_URL = "https://www.guangde.gov.cn"
LIST_TPL = BASE_URL + "/News/showList/933/page_%d.html"
SITE_NAME = "广德市人民政府-通知公告"
CATEGORY = "通知公告"
MAX_PAGES = 200  # 通知公告历史量级 (page_100 仍有 2022 数据)

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
TIMEOUT = 30

CUTOFF = "2020-01-01"


def parse_args():
    pages = 0
    for a in sys.argv[1:]:
        if a.startswith("--pages="):
            pages = int(a.split("=")[1])
        elif a.startswith("--pages") and len(sys.argv) > sys.argv.index(a) + 1:
            pages = int(sys.argv[sys.argv.index(a) + 1])
    return pages


def clean_title(t):
    t = html_mod.unescape(t or "")
    t = re.sub(r"[\u200b\u200e\u200f\ufeff\xa0]", "", t)
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"^[•·\-—]\s*", "", t)
    return t.strip()


def fetch(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if attempt == 2:
                print(f"  [WARN] fetch failed: {url} - {e}", file=sys.stderr)
                return ""
        time.sleep(1.5)
    return ""


def extract_list_items(html_text):
    """列表: <li><span>YYYY-MM-DD</span> <a href="/News/show/ID.html" title="完整标题">"""
    items = []
    soup = BeautifulSoup(html_text, "html.parser")
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        if "/News/show/" not in href:
            continue
        title = a.get("title") or a.get_text(strip=True)
        if not title:
            continue
        url = urljoin(BASE_URL, href)
        date = ""
        span = li.find("span")
        if span:
            m = re.search(r"(\d{4}-\d{2}-\d{2})", span.get_text())
            date = m.group(1) if m else ""
        items.append({"url": url, "title": clean_title(title), "date": date})
    # 去重
    seen = set()
    uniq = []
    for i in items:
        if i["url"] not in seen:
            seen.add(i["url"])
            uniq.append(i)
    return uniq


def clean_content(raw_html, page_url):
    """正文清洗: 保留 <p>/<table> HTML, 附件/图片绝对化"""
    # 附件/图片相对路径绝对化 (以详情页 URL 为基址)
    def _abs_url(u):
        if u.startswith(("http://", "https://", "javascript:", "mailto:", "tel:", "data:")):
            return u
        return urljoin(page_url, u)

    soup = BeautifulSoup(raw_html, "html.parser")
    # 剥 script/style
    for tag in soup.find_all(["script", "style"]):
        tag.decompose()
    # 附件/图片绝对化
    for a in soup.find_all("a", href=True):
        a["href"] = _abs_url(a["href"])
    for img in soup.find_all("img", src=True):
        img["src"] = _abs_url(img["src"])
    # 剥 二维码/分享/打印 容器
    for sel in [".ewm", ".ewm-1", ".ewm-2", ".share", ".print", ".qr", ".weixin-share"]:
        for el in soup.select(sel):
            el.decompose()
    # 规范化 <p> 标签
    for p in soup.find_all("p"):
        p.attrs = {}
    # 保留结构: table/tr/td 去属性, 其他标签剥除
    for tag in soup.find_all(True):
        if tag.name in ("table", "tbody", "thead", "tr", "td", "th", "p", "div", "span", "br", "a", "img"):
            if tag.name not in ("a", "img"):
                tag.attrs = {}
        else:
            tag.unwrap()
    # 空段落删除
    for p in soup.find_all("p"):
        if not p.get_text(strip=True) and not p.find("img") and not p.find("a"):
            p.decompose()
    # 序列化
    body = soup.decode_contents()
    body = re.sub(r"[\u3000\xa0]", " ", body)
    body = re.sub(r"\r\n", "\n", body)
    return body.strip()


def extract_detail(html_text, url):
    """详情: 标题/日期/正文/附件"""
    soup = BeautifulSoup(html_text, "html.parser")

    # 标题: meta ArticleTitle
    title = ""
    m = re.search(r'ArticleTitle"\s+content="([^"]*)"', html_text)
    if m:
        title = clean_title(m.group(1))

    # 日期: meta PubDate
    pub_date = ""
    m = re.search(r'PubDate"\s+content="([^"]*)"', html_text)
    if m:
        pm = re.search(r"(\d{4}-\d{2}-\d{2})", m.group(1))
        pub_date = pm.group(1) if pm else ""

    # 正文: div#zoom.detail_content
    body = ""
    zoom = soup.find("div", id="zoom")
    if zoom:
        body = clean_content(str(zoom), url)

    # 附件: 正文内 <a href="...pdf/doc...">
    attachments = []
    if body:
        for am in re.finditer(r'<a\s+href="([^"]+)"[^>]*>([^<]*)</a>', body):
            ahref, atext = am.group(1), am.group(2).strip()
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|7z)(\?|$)", ahref, re.I) or "download" in ahref.lower():
                attachments.append({"name": atext or ahref.split("/")[-1], "url": ahref})

    return {"title": title, "pub_date": pub_date, "content": body, "attachments": attachments}


def main():
    pages = parse_args()
    if not pages:
        pages = 5

    all_items = []
    seen = set()
    for pg in range(1, pages + 1):
        list_url = LIST_TPL % pg
        html_text = fetch(list_url)
        if not html_text:
            print(f"[PAGE {pg}] fetch failed, stopping")
            break
        items = extract_list_items(html_text)
        if not items:
            print(f"[PAGE {pg}] 0 items, stopping")
            break
        new_items = [i for i in items if i["url"] not in seen]
        seen.update(i["url"] for i in items)
        all_items.extend(new_items)
        print(f"[PAGE {pg}] +{len(new_items)} items (total {len(all_items)})")
        time.sleep(1.0)

    print(f"\n列表共 {len(all_items)} 条, 开始抓详情...")

    valid = []
    for i, item in enumerate(all_items, 1):
        detail_html = fetch(item["url"])
        if not detail_html:
            print(f"  [{i}/{len(all_items)}] SKIP fetch: {item['title'][:40]}")
            continue
        d = extract_detail(detail_html, item["url"])
        if not d["content"]:
            print(f"  [{i}/{len(all_items)}] SKIP 空正文: {item['title'][:40]}")
            continue
        title = d["title"] or item["title"]
        pub_date = d["pub_date"] or item["date"]
        content = d["content"]
        att = d["attachments"]
        # 附件嵌入正文尾部 (若正文未含)
        if att:
            has_att = any(re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|7z)(\?|$)", a.get("href", ""), re.I)
                          for a in re.finditer(r'<a\s+href="([^"]+)"', content))
            if not has_att:
                parts = [f'<p><a href="{a["url"]}">{a["name"]}</a></p>' for a in att]
                content = content + "\n" + "\n".join(parts)
        valid.append({
            "site_name": SITE_NAME,
            "title": title,
            "pub_date": pub_date,
            "content": content,
            "source_url": item["url"],
            "url": item["url"],
            "attachments": json.dumps(att, ensure_ascii=False) if att else "[]",
            "industry": "",
        })
        if i % 10 == 0:
            print(f"  [{i}/{len(all_items)}] 已处理...")
        time.sleep(0.5)

    print(f"\n有效 {len(valid)} 条, 入库...")
    push_to_searchdb(valid, CATEGORY)
    print(f"完成! 新增/更新 {len(valid)} 条")


if __name__ == "__main__":
    main()
