#!/usr/bin/env python3
"""crawl_yangxin_tzgg.py — 阳信县 通知公告 (yangxin.gov.cn col118189)"""
import re, sys, time, json, xml.etree.ElementTree as ET
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.yangxin.gov.cn"
SITE_NAME = "阳信县人民政府-通知公告"
COLUMN_NAME = "通知公告"
COLID = "118189"
UNITID = "694854"
WEBID = "451"
PER_PAGE = 15

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
}
session = requests.Session()
session.headers.update(HEADERS)


def fetch_list_range(start, end):
    url = f"{BASE_URL}/module/web/jpage/dataproxy.jsp"
    params = {
        "startrecord": str(start), "endrecord": str(end), "perpage": str(PER_PAGE),
        "unitid": UNITID, "webid": WEBID, "path": BASE_URL + "/",
        "col": "1", "columnid": COLID, "sourceContentType": "1",
        "webname": "%E9%98%B3%E4%BF%A1%E5%8E%BF%E4%BA%BA%E6%B0%91%E6%94%BF%E5%BA%9C",
        "permissiontype": "0",
    }
    data = {
        "col": "1", "webid": WEBID, "path": BASE_URL + "/",
        "columnid": COLID, "sourceContentType": "1", "unitid": UNITID,
        "webname": "%E9%98%B3%E4%BF%A1%E5%8E%BF%E4%BA%BA%E6%B0%91%E6%94%BF%E5%BA%9C",
        "permissiontype": "0",
    }
    try:
        resp = session.post(url, params=params, data=data, timeout=30,
                            headers={"Referer": f"{BASE_URL}/col/col{COLID}/index.html",
                                     "X-Requested-With": "XMLHttpRequest"})
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"  \u274c \u5217\u8868\u8bf7\u6c42\u5931\u8d25: {e}")
        return None


def parse_list_xml(xml_text):
    items = []
    try:
        root = ET.fromstring(xml_text)
        for record in root.findall(".//record"):
            cdata = record.text or ""
            a_match = re.search(r'href=["\']([^"\']+)["\']', cdata)
            title_match = re.search(r'title=["\']([^"\']+)["\']', cdata)
            date_match = re.search(r'<span>([^<]+)</span>', cdata)
            if a_match and title_match:
                href = a_match.group(1)
                if not href.startswith("http://www.yangxin.gov.cn/art/"):
                    continue
                items.append({
                    "title": title_match.group(1),
                    "url": href,
                    "date": date_match.group(1) if date_match else "",
                })
    except ET.ParseError as e:
        print(f"  \u274c XML\u89e3\u6790\u5931\u8d25: {e}")
    return items


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")
    title = ""
    h1 = soup.select_one("div.artcle h1")
    if h1:
        title = h1.get_text(strip=True)
    date = ""
    m = re.search(r"\u53d1\u5e03\u65e5\u671f[\uff1a:]\s*(\d{4}-\d{2}-\d{2})", html)
    if m:
        date = m.group(1)
    if not date:
        fbsj = soup.find("div", class_="fbsj")
        if fbsj:
            m2 = re.search(r"(\d{4}-\d{2}-\d{2})", fbsj.get_text())
            if m2:
                date = m2.group(1)
    content = ""
    attachments = []
    content_div = soup.find("div", class_="content")
    if content_div:
        parts = []
        processed = set()
        for el in content_div.find_all(["p", "table", "img", "h1", "h2", "h3", "h4"], recursive=True):
            if id(el) in processed:
                continue
            processed.add(id(el))
            if el.name == "table":
                md = table_to_markdown(el)
                if md:
                    parts.append(md)
            elif el.name == "img":
                src = el.get("src", "")
                alt = el.get("alt", "")
                if src:
                    # Skip decorative icons
                    if "/module/jslib/icons/" in src:
                        continue
                    full_src = urljoin(url, src) if not src.startswith("http") else src
                parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
            elif el.name == "p":
                txt = el.get_text(" ", strip=True)
                if txt:
                    txt = re.sub(r"[^\S\n]+", " ", txt)
                    parts.append(txt)
            else:
                txt = el.get_text(" ", strip=True)
                if txt:
                    parts.append(txt)
        content = "\n\n".join(parts)
        for a_tag in content_div.find_all("a", href=True):
            href = a_tag["href"]
            if "/module/download/downfile.jsp" in href or re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", href, re.I):
                fname = a_tag.get_text(strip=True) or href.split("/")[-1]
                full_url = urljoin(BASE_URL, href)
                attachments.append({"name": fname, "url": full_url})
    dowload = soup.find("div", class_="dowload")
    if dowload:
        for a_tag in dowload.find_all("a", href=True):
            href = a_tag["href"]
            fname = a_tag.get_text(strip=True) or href.split("/")[-1]
            full_url = urljoin(BASE_URL, href)
            if not any(a["url"] == full_url for a in attachments):
                attachments.append({"name": fname, "url": full_url})
    return title, content, date, attachments


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def main():
    import argparse
    parser = argparse.ArgumentParser(description="阳信县通知公告爬虫 (col118189)")
    parser.add_argument("--pages", type=int, default=5)
    args = parser.parse_args()

    sys.path.insert(0, "/root/gov_crawler")
    from crawler_lib import push_to_searchdb

    total = 0
    batch_count = (args.pages + 2) // 3
    for batch_idx in range(batch_count):
        start_record = batch_idx * 3 * PER_PAGE + 1
        end_record = min(args.pages * PER_PAGE, start_record + 3 * PER_PAGE - 1)
        print(f"[\u6279\u6b21 {batch_idx+1}/{batch_count}] start={start_record} end={end_record}")
        xml_text = fetch_list_range(start_record, end_record)
        if not xml_text:
            continue
        items = parse_list_xml(xml_text)
        if not items:
            print("  \u26a0\ufe0f \u65e0\u5217\u8868\u6570\u636e")
            continue
        seen_urls = set()
        unique_items = []
        for i, item in enumerate(items):
            record_num = start_record + i
            page_num = (record_num - 1) // PER_PAGE + 1
            if page_num > args.pages:
                continue
            if item["url"] not in seen_urls:
                seen_urls.add(item["url"])
                unique_items.append(item)
        print(f"  \u5171 {len(unique_items)} \u6761\u5f85\u722c")
        for item in unique_items:
            detail_url = item["url"]
            print(f"  \u2192 {item['title'][:40]}...", end=" ", flush=True)
            try:
                resp = session.get(detail_url, timeout=30)
                resp.encoding = "utf-8"
            except Exception as e:
                print(f"\u274c {e}")
                continue
            det_title, content, det_date, attachments = extract_detail(resp.text, detail_url)
            use_title = det_title or item["title"]
            use_date = det_date or item["date"]
            summary = content[:200].replace("\n", " ") if content else use_title
            if attachments:
                links = "\n\n**\u9644\u4ef6\uff1a**\n"
                for att in attachments:
                    links += f"- [{att['name']}]({att['url']})\n"
                content += links
            push_to_searchdb(
                items=[{
                    "title": use_title, "content": content, "summary": summary,
                    "url": detail_url, "site_name": SITE_NAME,
                    "pub_date": use_date,
                    "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
                }]
            )
            total += 1
            print(f"\u2705 {len(content)}\u5b57" if content else "\u26a0\ufe0f \u7a7a")
            time.sleep(0.3)
        time.sleep(0.5)
    print(f"\n=== \u5b8c\u6210\uff01\u5171\u5165\u5e93 {total} \u6761 ===")


if __name__ == "__main__":
    sys.exit(main())
