#!/usr/bin/env python3
"""
Crawl 新邱区人民政府 - 便民通知
https://www.fxxq.gov.cn/channel/list/11520.html
CMS: Custom government CMS
List: <a title="完整标题" href="/content/{year}/{id}.html"><span class="time">MM-DD</span></a>
Pagination: /channel/list/11520_2.html ... _19.html (19 pages total)
Detail: <meta ArticleTitle> for title, "日期：YYYY-MM-DD" for date, .Detailcontent_inner for content
"""
import requests, json, re, os, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.fxxq.gov.cn"
LIST_URL = "https://www.fxxq.gov.cn/channel/list/11520.html"
OUTPUT_FILE = "/root/gov_crawler/output/fxxq.jsonl"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)
session = requests.Session()
session.headers.update(HEADERS)

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_page_url(page_num):
    if page_num == 1:
        return LIST_URL
    return f"https://www.fxxq.gov.cn/channel/list/11520_{page_num}.html"

def get_total_pages(html):
    m = re.search(r"共(\d+)页", html)
    if m:
        return int(m.group(1))
    return 1

def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    list_container = soup.find("div", class_="list1")
    if not list_container:
        return items
    for a in list_container.find_all("a", href=True):
        href = a.get("href", "")
        if "/content/" not in href or "fuxin.gov.cn" in href:
            continue
        if not href.startswith("http"):
            href = urljoin(BASE_URL, href)
        # Title from div.list1-ul-link-inner inside a
        title_div = a.find("div", class_="list1-ul-link-inner")
        title = title_div.get_text(strip=True) if title_div else a.get_text(strip=True)
        if not title:
            continue
        # Date from sibling span.list1-ul-date
        date = ""
        parent_line = a.find_parent("div", class_="list1-line")
        if parent_line:
            span = parent_line.find("span", class_="list1-ul-date")
            if span:
                date = span.get_text(strip=True)
        items.append({"url": href, "title": title, "date": date})
    return items

def parse_detail(url, html):
    soup = BeautifulSoup(html, "html.parser")
    
    # Title from meta ArticleTitle
    title = ""
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()
    
    # Date from "日期：YYYY-MM-DD"
    pub_date = ""
    for tag in soup.find_all(["span", "div", "p", "td"]):
        txt = tag.get_text(strip=True)
        m = re.search(r"日期[：:]?\s*(\d{4}-\d{1,2}-\d{1,2})", txt)
        if m:
            pub_date = m.group(1)
            break
    
    # Content from .Detailcontent_inner
    content_div = soup.find("div", class_=re.compile(r"Detailcontent_inner"))
    if not content_div:
        content_div = soup.find("div", class_="news_inner")
    
    content_html = ""
    summary = ""
    attachments = []
    if content_div:
        for tag in content_div.find_all(True):
            attrs_to_keep = ["href", "src", "alt", "target", "border"]
            for attr in list(tag.attrs):
                if attr not in attrs_to_keep and not attr.startswith("mso-"):
                    del tag[attr]
        content_html = str(content_div)
        summary = body_text(content_div)
        for a in content_div.find_all("a", href=True):
            ahref = a.get("href", "")
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", ahref.lower()):
                full_url = ahref if ahref.startswith("http") else urljoin(url, ahref)
                attachments.append({"url": full_url, "text": a.get_text(strip=True) or os.path.basename(ahref)})
    
    return {
        "title": title,
        "pub_date": pub_date,
        "content_html": content_html,
        "summary": summary[:500] if summary else "",
        "attachments": attachments,
    }

def crawl_all():
    print("Fetching page 1...")
    resp = session.get(LIST_URL, timeout=30)
    resp.encoding = "utf-8"
    total_pages = get_total_pages(resp.text)
    print(f"Total pages: {total_pages}")
    
    all_items = []
    for page in range(1, total_pages + 1):
        url = get_page_url(page)
        print(f"  Page {page}/{total_pages}")
        try:
            if page == 1:
                resp_html = resp.text
            else:
                resp = session.get(url, timeout=30)
                resp.encoding = "utf-8"
                resp_html = resp.text
            items = parse_list(resp_html)
            print(f"    Found {len(items)} items")
            all_items.extend(items)
        except Exception as e:
            print(f"    [ERROR] {e}")
        time.sleep(0.3)
    
    # Dedup by URL
    seen = set()
    unique = []
    for item in all_items:
        if item["url"] not in seen:
            seen.add(item["url"])
            unique.append(item)
    print(f"\nTotal unique items: {len(unique)}")
    
    record_count = 0
    with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
        for i, item in enumerate(unique):
            preview = item["title"][:40] + "..." if len(item["title"]) > 40 else item["title"]
            print(f"  [{i+1}/{len(unique)}] {preview}")
            try:
                resp = session.get(item["url"], timeout=30)
                resp.encoding = "utf-8"
                detail = parse_detail(item["url"], resp.text)
                final_title = detail["title"] or item["title"]
                
                # Build full date: use detail date if available, else prepend year from URL
                date = detail["pub_date"]
                if not date:
                    m = re.search(r"/content/(\d{4})/", item["url"])
                    if m:
                        year = m.group(1)
                        if item["date"]:
                            date = f"{year}-{item['date'].replace('/', '-')}"
                
                record = {
                    "title": final_title,
                    "url": item["url"],
                    "date": date,
                    "content": detail["content_html"],
                    "summary": detail["summary"],
                    "site_name": "新邱区人民政府",
                    "group": "新邱区",
                    "attachments": json.dumps(detail["attachments"], ensure_ascii=False) if detail["attachments"] else "",
                }
                f.write(json.dumps(record, ensure_ascii=False) + "\n")
                record_count += 1
            except Exception as e:
                print(f"    [ERROR] {item['url']}: {e}")
            time.sleep(0.3)
    
    print(f"\nDone! {record_count} records written to {OUTPUT_FILE}")

if __name__ == "__main__":
    if "--incremental" in sys.argv:
        print("Incremental mode: crawl page 1 only")
        resp = session.get(LIST_URL, timeout=30)
        resp.encoding = "utf-8"
        items = parse_list(resp.text)
        print(f"Found {len(items)} items on page 1")
        with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
            for item in items:
                try:
                    resp = session.get(item["url"], timeout=30)
                    resp.encoding = "utf-8"
                    detail = parse_detail(item["url"], resp.text)
                    final_title = detail["title"] or item["title"]
                    date = detail["pub_date"]
                    if not date:
                        m = re.search(r"/content/(\d{4})/", item["url"])
                        if m and item["date"]:
                            date = f"{m.group(1)}-{item['date'].replace('/', '-')}"
                    record = {
                        "title": final_title,
                        "url": item["url"],
                        "date": date,
                        "content": detail["content_html"],
                        "summary": detail["summary"],
                        "site_name": "新邱区人民政府",
                        "group": "新邱区",
                        "attachments": json.dumps(detail["attachments"], ensure_ascii=False) if detail["attachments"] else "",
                    }
                    f.write(json.dumps(record, ensure_ascii=False) + "\n")
                except Exception as e:
                    print(f"ERROR: {e}")
                time.sleep(0.3)
        print(f"Incremental done: {len(items)} records")
    else:
        crawl_all()
