#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
湛江市遂溪县城市管理和综合执法局 - 政府信息公开平台 工作动态 (suixi.gov.cn/zjsxcgj/gkmlpt)
CMS: gkmlpt Vue SPA API (同 cnts/fogang /gkmlpt/api/all/{categoryId})
列表: GET /gkmlpt/api/all/5971?page=N&pageSize=100  JSON (211条, 2020-05~)
详情: GET /gkmlpt/api/post/{id}  JSON (content=HTML, date=字符串)
外链: type="url" 条目 (微信文章 mp.weixin.qq.com) 抓不到正文 -> 标题+URL内嵌段
正文: .article-content / #zoom (HTML fallback)
用法: python3 crawl_suixi_cgj_gzdt.py [--pages=N] [--json out.jsonl]
"""
import re, sys, os, json, time, sqlite3, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = "遂溪县城市管理和综合执法局-工作动态"
BASE_URL = "http://www.suixi.gov.cn"
API_URL = "http://www.suixi.gov.cn/zjsxcgj/gkmlpt/api/all/5971"
DOMAIN = "www.suixi.gov.cn"
GROUP = "广东"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
           "Referer": "http://www.suixi.gov.cn/zjsxcgj/gkmlpt/index"}


def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)


def clean_title(t):
    t = re.sub(r"&nbsp;|&#160;|\u200b|\ufeff", "", t or "")
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"\.\.\.|…$", "", t).strip()
    return t


def fetch(url, timeout=30):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = "utf-8"
    return r.text


def fetch_list(page, page_size=100):
    try:
        r = requests.get(f"{API_URL}?page={page}&pageSize={page_size}", headers=HEADERS, timeout=30)
        j = r.json()
        return j.get("articles", []), int(j.get("classify", {}).get("post_count", 0))
    except Exception as e:
        log(f"API P{page} ERR: {e}")
        return [], 0


def extract_clean_text(content_html, page_url=None):
    """HTML -> 文本, 附件/链接保留内嵌URL (用户偏好: HTML可点击链接段)"""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True) or "附件"
        base = page_url or BASE_URL
        full_url = urljoin(base, href)
        a_tag.replace_with(f'<p><a href="{full_url}">{text}</a></p>')
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()
    parts = []
    for el in soup.find_all(['table', 'p']):
        if el.name == 'table':
            parts.append(str(el))
        elif el.name == 'p' and not el.find_parent('table'):
            t = el.get_text(separator='', strip=True)
            if t:
                parts.append(f'<p>{t}</p>')
    return '\n\n'.join(parts) if parts else content_html.strip()


def parse_detail_api(art_id, page_url):
    """优先 API 详情; 返回 (title, publish_date, content) 或 (None,None,None)"""
    try:
        r = requests.get(f"{BASE_URL}/zjsxcgj/gkmlpt/api/post/{art_id}", headers=HEADERS, timeout=30)
        j = r.json()
        title = clean_title(j.get("title", ""))
        d = j.get("date", "")
        pub = str(d)[:10] if d else ""
        content_html = j.get("content", "") or ""
        content = extract_clean_text(content_html, page_url) if content_html else ""
        return title, pub, content
    except Exception as e:
        log(f"  api detail {art_id} ERR: {e}")
        return None, None, None


def parse_detail_html(html, page_url):
    result = {"title": "", "publish_date": "", "content": ""}
    soup = BeautifulSoup(html, "html.parser")
    title_tag = soup.find("title")
    if title_tag:
        t = title_tag.get_text(strip=True)
        t = re.sub(r"[_\-—]\s*(工作动态|政府信息公开平台|通知公告)\s*$", "", t)
        result["title"] = t.strip()
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        result["publish_date"] = meta_date["content"][:10]
    content_div = soup.find("div", class_="article-content")
    if not content_div:
        content_div = soup.find("div", class_="content")
    if not content_div:
        content_div = soup.find("div", id="zoom")
    if content_div:
        result["content"] = extract_clean_text(str(content_div), page_url)
    return result


def crawl(pages=5, json_out=None):
    log(f"Crawl pages={pages} cutoff={CUTOFF}")
    items_all = []
    seen = set()
    total = 0
    for p in range(1, pages + 1):
        arts, total = fetch_list(p)
        new = 0
        for a in arts:
            aid = a.get("id")
            title = clean_title(a.get("title", ""))
            ts = a.get("date", 0)
            if ts:
                try:
                    pub_date = datetime.fromtimestamp(int(ts)).strftime("%Y-%m-%d")
                except (ValueError, TypeError, OSError):
                    pub_date = str(ts)[:10]
            else:
                pub_date = ""
            atype = a.get("type", "normal")
            # type=url 外链: 直接用列表里的 url 字段
            if atype == "url":
                url = a.get("url", "") or a.get("post_url", "") or ""
                if not url.startswith("http"):
                    url = urljoin(BASE_URL, url)
            else:
                url = f"{BASE_URL}/zjsxcgj/gkmlpt/content/post_{aid}.html"
            if url and url not in seen:
                seen.add(url)
                items_all.append({"title": title, "url": url, "pub_date": pub_date, "art_id": aid, "type": atype})
                new += 1
        log(f"P{p}: {len(arts)} items ({new} new), total {len(items_all)} (post_count={total})")
        if not arts:
            break
        time.sleep(0.6)

    enriched = []
    for idx, it in enumerate(items_all, 1):
        if it["pub_date"] and it["pub_date"] < CUTOFF:
            continue
        art_id = it.get("art_id")
        if it.get("type") == "url":
            # 纯外链 (微信文章): 正文=标题+URL内嵌段 (用户偏好)
            title = it["title"]
            url = it["url"]
            content = f'<p>{title}</p>\n<p><a href="{url}">查看原文</a></p>'
            enriched.append({
                "title": title,
                "page_url": url,
                "publish_date": it["pub_date"],
                "content": content,
                "site_name": SITE_NAME,
                "column": "工作动态",
            })
            if idx % 10 == 0:
                log(f"  {idx}/{len(items_all)}")
            continue
        try:
            title, pub, content = parse_detail_api(art_id, it["url"])
            if not content:
                html = fetch(it["url"])
                d2 = parse_detail_html(html, it["url"])
                title = title or d2["title"]
                pub = pub or d2["publish_date"]
                content = d2["content"]
        except Exception as e:
            log(f"  detail {it['url'][-40:]} ERR: {e}")
            title, pub, content = it["title"], it["pub_date"], ""
        enriched.append({
            "title": title or it["title"],
            "page_url": it["url"],
            "publish_date": pub or it["pub_date"],
            "content": content,
            "site_name": SITE_NAME,
            "column": "工作动态",
        })
        if idx % 10 == 0:
            log(f"  {idx}/{len(items_all)}")
        time.sleep(0.4)

    if json_out:
        with open(json_out, "w", encoding="utf-8") as f:
            for item in enriched:
                f.write(json.dumps(item, ensure_ascii=False) + "\n")
        log(f"JSONL: {len(enriched)} items -> {json_out}")
        return

    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    stored = skipped = 0
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category, script_name)"
                " VALUES (?,?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], item["column"], "crawl_suixi_cgj_gzdt.py"))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


if __name__ == "__main__":
    args = sys.argv[1:]
    pages = 5
    json_out = None
    if "--pages=" in " ".join(args):
        pages = int(re.search(r"--pages=(\d+)", " ".join(args)).group(1))
    if "--json" in args:
        json_out = args[args.index("--json") + 1]
    crawl(pages=pages, json_out=json_out)
