#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
第六师五家渠市人民政府 - 在线征集 (wjq.gov.cn/wykwybwyw/wyw/zxzj)
CMS: JPAAS 调查征集 (jsurvey-web-server)
列表: GET /api-gateway/jpaas-jsurvey-web-server/interface/dczj/find-title-list
      params: webId=gXiRWWCDbJQMHYm3xJLgF&pageSize=20&pageNo=N&orderType=0&type=1  JSON (79条/4页)
详情: GET /front/dczj/showJsurveys.do?formId={iid}  正文 td.jsurvey-td-xj(第3个, p段)
标题: 详情页 <title> | 日期: 列表 createTime
用法: python3 crawl_wjq_zxzj.py [--pages=N] [--json out.jsonl]
"""
import re, sys, os, json, time, sqlite3, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

SITE_NAME = "第六师五家渠市人民政府-在线征集"
BASE_URL = "https://www.wjq.gov.cn"
LIST_API = BASE_URL + "/api-gateway/jpaas-jsurvey-web-server/interface/dczj/find-title-list"
WEB_ID = "gXiRWWCDbJQMHYm3xJLgF"
DOMAIN = "www.wjq.gov.cn"
GROUP = "新疆"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
           "Referer": "https://www.wjq.gov.cn/wykwybwyw/wyw/zxzj/index.html"}
HEADERS_MOBILE = {"User-Agent": "Mozilla/5.0 (iPhone; CPU iPhone OS 16_0 like Mac OS X) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/16.0 Mobile/15E148 Safari/604.1",
                  "Referer": "https://www.wjq.gov.cn/wykwybwyw/wyw/zxzj/index.html"}


def log(msg):
    print(f"[{SITE_NAME}] {msg}", flush=True)


def clean_title(t):
    t = re.sub(r"&nbsp;|&#160;|\u200b|\ufeff", "", t or "")
    t = re.sub(r"\s+", " ", t)
    t = re.sub(r"\.\.\.|…$", "", t).strip()
    return t


def fetch_list(page, page_size=20):
    try:
        params = {"webId": WEB_ID, "pageSize": page_size, "pageNo": page,
                  "orderType": 0, "type": 1}
        r = requests.get(LIST_API, params=params, headers=HEADERS, timeout=30)
        j = r.json()
        if j.get("code") != "200" or not j.get("success"):
            log(f"API P{page} fail: {j.get('message')}")
            return [], 0
        d = j.get("data", {})
        return d.get("list", []), int(d.get("totalCount", 0))
    except Exception as e:
        log(f"API P{page} ERR: {e}")
        return [], 0


def extract_clean_text(content_html, page_url=None):
    """HTML -> 文本, 链接/附件保留内嵌URL (用户偏好: HTML可点击链接段)"""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True) or "附件"
        base = page_url or BASE_URL
        full_url = urljoin(base, href)
        a_tag.replace_with(f'<p><a href="{full_url}">{text}</a></p>')
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()
    parts = []
    for el in soup.find_all(['table', 'p']):
        if el.name == 'table':
            parts.append(str(el))
        elif el.name == 'p' and not el.find_parent('table'):
            t = el.get_text(separator='', strip=True)
            if t:
                parts.append(f'<p>{t}</p>')
    return '\n\n'.join(parts) if parts else content_html.strip()


def parse_detail(html, page_url):
    """提取正文: 优先手机版 td 内 p 段 (已结束征集手机版也显示正文)"""
    soup = BeautifulSoup(html, "html.parser")
    t = soup.find("title")
    title = clean_title(t.get_text(strip=True)) if t else ""
    # 去掉站点后缀
    title = re.sub(r"_\s*(在线征集|五家渠市人民政府)\s*$", "", title)
    # 日期
    pub = ""
    m = re.search(r"发布时间[:：]\s*(\d{4}-\d{2}-\d{2})", html)
    if m:
        pub = m.group(1)
    # 正文: 优先 td, 找不到再找 rwtpjgtop (问卷结束提示后的正文区)
    content = ""
    best = None
    best_len = 0
    for td in soup.find_all("td"):
        txt = td.get_text(strip=True)
        # 排除纯表单/导航 td
        if any(k in txt for k in ["提交成功", "当前共投了", "问卷已结束", "结果生成时间", "共计:票"]):
            if len(txt) < 200:
                continue
        if len(txt) > 100 and len(txt) > best_len:
            best = td
            best_len = len(txt)
    if best:
        content = extract_clean_text(str(best), page_url)
    else:
        # 问卷结束后的正文区: rwtpjgtop(display:block) 内的文本 (去掉投票/提示噪音)
        rwts = soup.find_all("div", class_="rwtpjgtop")
        for rwt in rwts:
            style = rwt.get("style", "")
            if "display:block" in style.replace(" ", "") or "display: block" in style:
                body = BeautifulSoup(str(rwt), "html.parser")
                # 去掉投票结果/提示标签
                for junk in body.find_all(["p", "img"]):
                    t = junk.get_text(strip=True)
                    if t in ("提交成功", "问卷结束") or "当前共投了" in t or "结果生成时间" in t or "问卷已结束，感谢您的参与！" in t or "结束时间：" in t:
                        junk.decompose()
                for tag in body.find_all(["div", "span", "b", "strong", "font", "em", "i", "u", "s"]):
                    tag.unwrap()
                parts = []
                for el in body.find_all("p"):
                    t = el.get_text(separator="", strip=True)
                    if t:
                        parts.append(f"<p>{t}</p>")
                content = "\n\n".join(parts)
                if content:
                    break
    return title, pub, content


def crawl(pages=5, json_out=None):
    log(f"Crawl pages={pages} cutoff={CUTOFF}")
    items_all = []
    seen = set()
    total = 0
    for p in range(1, pages + 1):
        arts, total = fetch_list(p)
        new = 0
        for a in arts:
            title = clean_title(a.get("title", ""))
            ct = a.get("createTime", "")
            pub = ct[:10] if ct else ""
            url = a.get("url", "")
            iid = a.get("iid", "")
            state = a.get("state", 0)
            if url and url not in seen:
                seen.add(url)
                items_all.append({"title": title, "url": url, "pub_date": pub, "iid": iid, "state": state})
                new += 1
        log(f"P{p}: {len(arts)} items ({new} new), total {len(items_all)} (totalCount={total})")
        if not arts:
            break
        time.sleep(0.6)

    enriched = []
    for idx, it in enumerate(items_all, 1):
        if it["pub_date"] and it["pub_date"] < CUTOFF:
            continue
        try:
            # 已结束(state=2)条目 PC 版无正文 -> 用手机版
            if it.get("state") == 2:
                mob = re.sub(r"showJsurveys\.do", "showPhoneJsurveys.do", it["url"])
                r = requests.get(mob, headers=HEADERS_MOBILE, timeout=30)
            else:
                r = requests.get(it["url"], headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            html = r.text
            title, pub, content = parse_detail(html, it["url"])
        except Exception as e:
            log(f"  detail ERR: {e}")
            title, pub, content = it["title"], it["pub_date"], ""
        enriched.append({
            "title": title or it["title"],
            "page_url": it["url"],
            "publish_date": pub or it["pub_date"],
            "content": content,
            "site_name": SITE_NAME,
            "column": "在线征集",
        })
        if idx % 10 == 0:
            log(f"  {idx}/{len(items_all)}")
        time.sleep(0.4)

    if json_out:
        with open(json_out, "w", encoding="utf-8") as f:
            for item in enriched:
                f.write(json.dumps(item, ensure_ascii=False) + "\n")
        log(f"JSONL: {len(enriched)} items -> {json_out}")
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    stored = skipped = 0
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category, script_name)"
                " VALUES (?,?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], item["column"], "crawl_wjq_zxzj.py"))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


if __name__ == "__main__":
    args = sys.argv[1:]
    pages = 5
    json_out = None
    if "--pages=" in " ".join(args):
        pages = int(re.search(r"--pages=(\d+)", " ".join(args)).group(1))
    if "--json" in args:
        json_out = args[args.index("--json") + 1]
    crawl(pages=pages, json_out=json_out)
