#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
深圳市生态环境局 meeb.sz.gov.cn — 环保业务栏目 (xxgk/qt/hbyw)
CMS: NFCMS (广东南方网系) + 广东省统一搜索平台 JSONP API
数据源: https://search.gd.gov.cn/jsonp/site/755014 (site=755014 = meeb NFCMS_SITE_ID)
  ?callback=datacallback&category_id=24525,...,24514&order=1&text=&pagesize=15&page=N&including_link_doc=1
列表: JSONP 直接返回 {count, results:[{title,url,post_url,pub_time,content(截断摘要),attachment:[...]}]}
详情: curl 详情页 → meta ArticleTitle / article_info 发布日期 / div.article 正文 / div.article_pdf 附件
注意: --curves prime256v1 必须 (服务器 TLS 曲线协商失败 bad ecpoint)
count=407, 15/页=28 页; CUTOFF 边界 P13 (2023-08-04 末条 < 2023-08-09)
"""
import argparse, json, os, re, sqlite3, subprocess, sys, time
from datetime import datetime, timedelta
from urllib.parse import urljoin

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
SITE_NAME = "深圳市生态环境局-环保业务"
CATEGORY_ID = "24525,24524,24523,24522,24521,24520,24519,24518,24517,24516,24515,24514"
API_URL = "https://search.gd.gov.cn/jsonp/site/755014"
DB_PATH = "/root/search.db"  # 软链到 /mnt/data/search.db

CUTOFF = (datetime.now() - timedelta(days=365 * 3)).strftime("%Y-%m-%d")
print(f"[i] CUTOFF 边界: {CUTOFF}", file=sys.stderr)


def curl_get(url, referer=None, cj=None, timeout=30):
    cmd = ["curl", "-sk", "--curves", "prime256v1", "-m", str(timeout), "-A", UA]
    if referer:
        cmd += ["-H", f"Referer: {referer}"]
    if cj:
        cmd += ["-c", cj, "-b", cj]
    cmd += ["-L", url]
    for attempt in range(3):
        try:
            r = subprocess.run(cmd, capture_output=True, text=True, timeout=timeout + 10)
            if r.stdout:
                return r.stdout
        except Exception as e:
            print(f"[!] curl err {e}", file=sys.stderr)
        time.sleep(2)
    return ""


def fetch_jsonp(page):
    url = f"{API_URL}?callback=datacallback&category_id={CATEGORY_ID}&order=1&text=&pagesize=15&page={page}&including_link_doc=1"
    out = curl_get(url)
    m = re.search(r"^[^(]*\((.*)\)\s*$", out, re.S)
    if not m:
        return None
    try:
        return json.loads(m.group(1))
    except Exception:
        return None


def parse_detail(url):
    """返回 (title, pub_date, content, attachments)"""
    h = curl_get(url, referer="https://meeb.sz.gov.cn/xxgk/qt/hbyw/")
    if not h:
        return None, None, "", []
    # 标题: meta ArticleTitle
    title = ""
    mt = re.search(r'<meta name="ArticleTitle" content="([^"]+)"', h)
    if mt:
        title = mt.group(1).strip()
    if not title:
        mt = re.search(r"<title>(.*?)</title>", h, re.S)
        if mt:
            title = mt.group(1).strip().replace("_深圳市生态环境局", "").replace("-深圳市生态环境局", "").strip()
    # 日期: article_info 发布日期
    pub_date = ""
    dm = re.search(r"发布日期：(\d{4}-\d{2}-\d{2})", h)
    if dm:
        pub_date = dm.group(1)
    # 正文: div.article (到 article_pdf 为止)
    content = ""
    m = re.search(r'<div class="article">(.*?)<div class="article_pdf"', h, re.S)
    if not m:
        m = re.search(r'<div class="article">(.*?)</div>\s*<div class="article_push"', h, re.S)
    if m:
        seg = m.group(1)
        # 图片型公告: 正文无文本但有图 → 保留原图
        imgs = [u for u in re.findall(r'<img[^>]+src="([^"]+)"', seg)]
        # 保留 <p> 段落 HTML（search_app 检测到 <p>/<table> 直接当 HTML 渲染，拍平会挤成一坨）
        # 段落内: 剥内层标签、全角空格→普通空格; 段落间保留 <p> 分隔
        content = ""
        for p in re.findall(r'<p[^>]*>(.*?)</p>', seg, re.S):
            txt = re.sub(r"<[^>]+>", "", p)
            txt = re.sub(r"[ \t\u3000]+", " ", txt)
            txt = re.sub(r"\s+", " ", txt)
            txt = txt.strip()
            if txt:
                content += f"<p>{txt}</p>\n"
        # 无 <p> 的裸文本/段落
        if not content:
            txt = re.sub(r"<[^>]+>", "", seg)
            txt = re.sub(r"[ \t\u3000]+", " ", txt)
            txt = re.sub(r"\s+", " ", txt)
            txt = txt.strip()
            if txt:
                content = f"<p>{txt}</p>"
        if not content and imgs:
            for src in imgs:
                if not src.startswith("http"):
                    src = urljoin("https://meeb.sz.gov.cn", src)
                content += f'<p><img src="{src}"></p>\n'
            content = content.strip()
    # 附件: article_pdf 内 a.file/a.doc
    attachments = []
    m = re.search(r'<div class="article_pdf">(.*?)</div>', h, re.S)
    if m:
        for a in re.finditer(r'<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>', m.group(1), re.S):
            href = a.group(1).strip()
            name = re.sub(r"<[^>]+>", "", a.group(2)).strip()
            if not name:
                name = os.path.basename(href)
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|et|txt|jpg|png)$", href, re.I):
                if not href.startswith("http"):
                    href = urljoin("https://meeb.sz.gov.cn", href)
                attachments.append(f'<p><a href="{href}">{name}</a></p>')
    return title, pub_date, content, attachments


def push_to_searchdb(records):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    cur.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, publish_date TEXT, content TEXT,
        page_url TEXT, source_url TEXT, site_name TEXT,
        summary TEXT, date_rank INTEGER,
        UNIQUE(page_url, site_name))""")
    new = skip = 0
    for rec in records:
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw (title, publish_date, content, page_url, source_url, site_name, summary, date_rank) VALUES (?,?,?,?,?,?,?,?)",
            (rec["title"], rec["pub_date"], rec["content"], rec["url"], rec["url"], SITE_NAME,
             rec["content"][:200] if rec["content"] else "", rec["date_rank"]),
        )
        if cur.rowcount:
            new += 1
        else:
            skip += 1
    conn.commit()
    conn.close()
    return new, skip


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--pages", type=int, default=5)
    ap.add_argument("--start", type=int, default=1)
    args = ap.parse_args()

    total_new = total_skip = 0
    for pg in range(args.start, args.start + args.pages):
        d = fetch_jsonp(pg)
        if not d or not d.get("results"):
            print(f"[P{pg}] empty -> stop", file=sys.stderr)
            break
        results = d["results"]
        print(f"[P{pg}] {len(results)} 条 (count={d.get('count')})", file=sys.stderr)
        records = []
        for it in results:
            url = it.get("post_url") or it.get("url") or ""
            pub = (it.get("pub_time") or "")[:10]
            if not pub or not url:
                continue
            if pub < CUTOFF:
                print(f"[P{pg}] 跳过窗口外 {pub} {it.get('title','')[:30]}", file=sys.stderr)
                continue
            # API content 是截断摘要，必须抓详情页
            title, dpub, content, atts = parse_detail(url)
            if not title:
                title = it.get("title", "").strip()
            if not dpub:
                dpub = pub
            if content and atts:
                content = content + "\n\n" + "\n".join(atts)
            elif atts:
                content = "\n".join(atts)
            records.append({
                "title": title, "pub_date": dpub, "content": content, "url": url,
                "date_rank": int(datetime.strptime(dpub, "%Y-%m-%d").timestamp()),
            })
            time.sleep(0.8)  # 详情页请求间隔
        if records:
            n, s = push_to_searchdb(records)
            total_new += n
            total_skip += s
            print(f"[P{pg}] new={n} skip={s}", file=sys.stderr)
        time.sleep(1.0)  # 列表页间隔
    print(f"DONE new={total_new} skip={total_skip}")


if __name__ == "__main__":
    main()
