#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""永丰县人民政府 - 公告公示栏目爬虫
站点: http://www.jxyongfeng.gov.cn (江西吉安永丰县, 维网科技 CMS)
栏目: news-list-gggsaa.html (公告公示, catid=4734)
WAF: 无 (requests 直连 HTTP 200)
列表: POST /api-ajax_list-{page}.html  jQuery 数组序列化 body:
      ajax_type[]=5_news&ajax_type[]=4734&ajax_type[]=5&ajax_type[]=news&ajax_type[]=Y-m-d&ajax_type[]=40&ajax_type[]=20&ajax_type[]=["is_top DESC","displayorder DESC","inputtime DESC"]&ajax_type[]=&is_ds=1
      返回 JSON {total, data:[{id,title,url,inputtime,...}]}, 20条/页
详情: /news-show-{id}.html  UTF-8 (必须 r.content.decode('utf-8'), 不能用 r.text 否则 latin-1 乱码)
      正文 div.xxq_essay (id=content); 标题 meta ArticleTitle; 日期 meta PubDate
      纯PDF型: iframe src=/uploadfile/5/Attachment/xxx.pdf → 正文=标题+URL内嵌段
      附件: <a href="/uploadfile/...">文件名.xlsx</a> 相对路径 → BASE 绝对化
入库: 自动检测 /root/search.db 存在则 INSERT OR IGNORE 直插 gov_raw (FTS 触发器自动同步 gov_search)
运行: python3 crawl_yongfeng_gggs.py [--pages=N] [--no-db]
"""
import json, os, re, sys, time, sqlite3, html as H
from datetime import datetime, timedelta
import requests

SITE_NAME = "永丰县-公告公示"
BASE = "http://www.jxyongfeng.gov.cn"
CATID = "4734"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
OUT = "/tmp/yongfeng_gggs.jsonl"
DB_PATH = "/root/search.db"
CATEGORY = "公告公示"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": BASE + "/news-list-gggsaa.html",
    "X-Requested-With": "XMLHttpRequest",
}

def clean_title(t):
    t = re.sub(r"^[•··.\s]+", "", t or "").strip()
    t = t.replace("\u200b", "").replace("\ufeff", "")
    return t

def clean_content(html):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    for tag in soup.find_all(lambda t: t.name and ":" in t.name):
        tag.decompose()
    for tag in soup.find_all(["span", "font"]):
        tag.unwrap()
    # 只提升含嵌套 p 的外层 p (如 <p 外壳><div><p>段1</p><p>段2</p></div></p>),
    # 保留内层真段落 p — 勿 unwrap 内层 p 否则段落被拍平成一坨
    while True:
        nested = [p for p in soup.find_all("p") if p.find("p")]
        if not nested:
            break
        nested[0].unwrap()
    # 提升正文容器 div (vw_editor 等), 段落直接裸露
    for div in soup.find_all("div", class_="vw_editor"):
        div.unwrap()
    return str(soup)

def absolutize(href):
    if not href or href.startswith("javascript"):
        return href
    href = href.strip()
    if href.startswith("http"):
        return href.replace("http://", "http://")
    if href.startswith("/"):
        return BASE + href
    return BASE + "/" + href

def extract_attachments(html_frag):
    atts = []
    for m in re.finditer(r'<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>', html_frag, re.I | re.S):
        href, name_html = m.group(1).strip(), m.group(2)
        if not href or href.startswith("javascript"):
            continue
        if re.search(r'\.(pdf|docx?|xlsx?|zip|rar|7z)$', href, re.I):
            name = H.unescape(re.sub(r"<[^>]+>", "", name_html)).strip()[:100]
            atts.append({"name": name or href.split("/")[-1], "url": absolutize(href)})
    seen = set()
    uniq = []
    for a in atts:
        if a["url"] not in seen:
            seen.add(a["url"])
            uniq.append(a)
    return uniq

def fetch_page(page):
    data = {
        "ajax_type[]": ["5_news", CATID, 5, "news", "Y-m-d", 40, 20,
                        ['is_top DESC', 'displayorder DESC', 'inputtime DESC'], ""],
        "is_ds": 1,
    }
    for attempt in range(3):
        try:
            r = requests.post(f"{BASE}/api-ajax_list-{page}.html", data=data, headers=HEADERS, timeout=30)
            if r.status_code == 200:
                return r.json()
        except Exception as e:
            print(f"  API第{page}页失败({attempt+1}): {e}", flush=True)
            time.sleep(2)
    return None

def fetch_detail(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            if r.status_code == 200:
                return r.content.decode("utf-8", errors="replace")
        except Exception as e:
            print(f"  详情失败({attempt+1}): {e}", flush=True)
            time.sleep(2)
    return None

def push_to_db(results):
    if not results:
        return 0, 0
    if not os.path.exists(DB_PATH):
        print(f"  未发现 {DB_PATH}, 跳过入库 (仅写 JSONL)", flush=True)
        return 0, 0
    conn = sqlite3.connect(DB_PATH, timeout=290)
    conn.execute("PRAGMA busy_timeout=290000")
    added = skipped = 0
    cur = conn.cursor()
    for r in results:
        try:
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name, attachments, group_name, industry)
                   VALUES (?,?,?,?,?,?,?,?,?,?,?,?)""",
                (SITE_NAME, r["url"], r["url"], r["title"], r["pub_date"], r["content"],
                 (r["content"] or "")[:500], CATEGORY, "crawl_yongfeng_gggs.py", r["attachments"],
                 r.get("group_name", "江西"), r.get("industry", "other"))
            )
            if cur.rowcount > 0:
                added += 1
            else:
                skipped += 1
        except sqlite3.Error as e:
            print(f"  DB err: {e}", flush=True)
            skipped += 1
        conn.commit()
    conn.close()
    return added, skipped

def main():
    no_db = "--no-db" in sys.argv
    max_pages = None
    for a in sys.argv:
        if a.startswith("--pages="):
            max_pages = int(a.split("=", 1)[1])
    print(f"站点: {SITE_NAME} CUTOFF: {CUTOFF}", flush=True)

    results = []
    stats = {"list": 0, "detail_ok": 0, "empty": 0, "cutoff": 0, "total": 0, "pages": 0}
    cutoff_hit = False
    n = 1
    while True:
        if max_pages and n > max_pages:
            break
        j = fetch_page(n)
        if not j or not j.get("total"):
            print(f"  第{n}页无数据, 停止", flush=True)
            break
        if stats["total"] == 0:
            stats["total"] = j["total"]
            stats["pages"] = (j["total"] + 19) // 20
            print(f"共 {stats['total']} 条 / {stats['pages']} 页", flush=True)
        items = j.get("data", [])
        if not items:
            print(f"  第{n}页无条目, 停止", flush=True)
            break
        print(f"  第{n}页: {len(items)} 条", flush=True)
        for it in items:
            stats["list"] += 1
            list_title = clean_title(it.get("title", ""))
            pub_date = (it.get("inputtime") or "")[:10]
            if pub_date and pub_date < CUTOFF:
                stats["cutoff"] += 1
                cutoff_hit = True
                continue
            detail_url = it.get("url") or f"{BASE}/news-show-{it.get('id')}.html"
            detail_html = fetch_detail(detail_url)
            if not detail_html:
                print(f"    详情失败: {list_title[:30]}", flush=True)
                continue
            stats["detail_ok"] += 1
            m = re.search(r'<meta[^>]+name="ArticleTitle"[^>]+content="([^"]+)"', detail_html)
            title = clean_title(H.unescape(m.group(1))) if m else list_title
            m = re.search(r'<meta[^>]+name="PubDate"[^>]+content="([^"]+)"', detail_html)
            pub_date = (m.group(1) if m else pub_date)[:10]
            if pub_date and pub_date < CUTOFF:
                stats["cutoff"] += 1
                cutoff_hit = True
                continue
            m = re.search(r'<div class="xxq_essay"[^>]*>(.*?)<div class="xxq_prev"', detail_html, re.S)
            content_html = m.group(1) if m else ""
            # 纯PDF型: iframe 嵌入 PDF (正文无文字)
            iframes = re.findall(r'<iframe[^>]+src="([^"]+\.pdf[^"]*)"', content_html)
            text = re.sub(r"<[^>]+>", "", content_html).strip()
            if not content_html or (len(text) < 10 and not iframes):
                print(f"    空正文跳过: {title[:30]}", flush=True)
                stats["empty"] += 1
                continue
            if iframes:
                # 正文 = 标题 + URL 内嵌段 (纯PDF外链公告模式)
                pdf_url = absolutize(iframes[0])
                content = f'<p><a href="{pdf_url}">点击查看PDF公告</a></p><p><a href="{pdf_url}">{pdf_url}</a></p>'
            else:
                content = clean_content(content_html)
                content = re.sub(
                    r'(<img[^>]+src=")([^"]+)(")',
                    lambda mm: mm.group(1) + absolutize(mm.group(2)) + mm.group(3), content)
            atts = extract_attachments(content_html)
            rec = {
                "site_name": SITE_NAME,
                "title": title,
                "pub_date": pub_date,
                "content": content,
                "source_url": detail_url,
                "url": detail_url,
                "group_name": "江西",
                "industry": "other",
                "attachments": json.dumps(atts, ensure_ascii=False) if atts else "",
            }
            results.append(rec)
        if cutoff_hit:
            break
        if n >= stats["pages"]:
            break
        n += 1

    with open(OUT, "w", encoding="utf-8") as f:
        for r in results:
            f.write(json.dumps(r, ensure_ascii=False) + "\n")
    print(f"已写 {len(results)} 条到 {OUT}", flush=True)
    if not no_db:
        added, skipped = push_to_db(results)
        print(f"=== 入库: 新增{added} 跳过{skipped} ===", flush=True)
    print(f"=== 完成: 列表{stats['list']} 详情OK{stats['detail_ok']} 空{stats['empty']} CUTOFF截{stats['cutoff']} ===", flush=True)

if __name__ == "__main__":
    main()
