#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""苍南县人民政府 - 通知公告 crawler"""
import re, sys, os, time, sqlite3, json, requests

SITE_NAME = "苍南县人民政府-通知公告"
BASE_URL = "https://www.cncn.gov.cn"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
PER_PAGE = 20

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36",
    "Referer": "https://www.cncn.gov.cn/col/col1229416690/index.html",
}

def log(msg):
    print(f"[{time.strftime('%H:%M:%S')}] {msg}")

def fetch_list_page(page):
    params = {
        "webId": "1831",
        "pageId": "1229416690",
        "parseType": "bulidstatic",
        "pageType": "column",
        "tagId": "\u53f3\u4fa7\u5217\u8868",
        "tplSetId": "Mm0JzuVkej5cUObgIOwPx",
        "paramJson": json.dumps({"pageNo": page, "pageSize": PER_PAGE})
    }
    for attempt in range(3):
        try:
            r = requests.get(API_URL, params=params, headers=HEADERS, timeout=60)
            r.encoding = "utf-8"
            d = r.json()
            if d.get("success"):
                return d["data"]["html"]
            log(f"API error: {d.get('message', 'unknown')}")
            return None
        except Exception as e:
            log(f"请求失败(第{attempt+1}次): {str(e)[:50]}")
            if attempt < 2:
                time.sleep(3 * (attempt + 1))
    return None

def parse_articles(html):
    arts = []
    dn = chr(92) + "d"
    dpat = f'<span>({dn*4}-{dn*2}-{dn*2})</span>'
    for item in re.findall(r'<div class="news-item">(.*?)</div>', html, re.DOTALL):
        m = re.search(r'<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"', item)
        dm = re.search(dpat, item)
        if m:
            url = m.group(1).strip()
            if not url.startswith("http"):
                url = BASE_URL + url
            title = m.group(2).strip()
            pub_date = dm.group(1).strip() if dm else ""
            arts.append({"title": title, "url": url, "pub_date": pub_date})
    return arts

def fetch_detail(url):
    for attempt in range(2):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            html = r.text
            break
        except:
            if attempt == 0:
                time.sleep(2)
            else:
                return "", []
    zm = re.search(r'<div id="zoom">(.*?)</div>\s*', html, re.DOTALL)
    zh = zm.group(1) if zm else ""
    lines = []
    for p in re.findall(r'<p[^>]*>(.*?)</p>', zh, re.DOTALL):
        t = re.sub(r'<[^>]+>', '', p).strip().replace('&nbsp;', ' ').replace('\u3000', ' ').strip()
        if t:
            lines.append(t)
    content = "\n".join(lines)
    atts = []
    for m in re.finditer(r'<a[^>]*download="([^"]*)"[^>]*href="([^"]*)"', zh):
        nm = m.group(1).strip()
        lk = m.group(2)
        if not lk.startswith("http"):
            lk = BASE_URL + lk
        atts.append({"url": lk, "name": nm})
    return content, atts

def main():
    log("开始爬取苍南县通知公告...")
    html = fetch_list_page(1)
    if not html:
        log("ERROR: 第1页获取失败")
        return
    all_arts = parse_articles(html)
    log(f"第1页: {len(all_arts)} 条")
    if len(all_arts) == 0:
        log("ERROR: 解析到0条，退出")
        return
    total_pages = 148
    for p in range(2, total_pages + 1):
        time.sleep(0.5)
        h = fetch_list_page(p)
        if not h:
            log(f"第{p}页失败，跳过")
            continue
        arts = parse_articles(h)
        all_arts.extend(arts)
        if p % 20 == 0:
            log(f"第{p}页: {len(arts)} 条 (累计{len(all_arts)})")
    log(f"列表共 {len(all_arts)} 条")
    db = sqlite3.connect(SEARCH_DB, timeout=30)
    db.execute("PRAGMA journal_mode=WAL")
    existing = set()
    try:
        cur = db.execute("SELECT title FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        for row in cur.fetchall():
            existing.add(row[0])
    except:
        pass
    log(f"已有记录: {len(existing)} 条")
    saved, skipped = 0, 0
    for art in all_arts:
        if art["title"] in existing:
            skipped += 1
            continue
        try:
            content, atts = fetch_detail(art["url"])
        except:
            content, atts = "", []
        body = ""
        if content:
            body = "<p>" + content.replace("\n", "</p><p>") + "</p>"
        for a in atts:
            body += f'<p><a href="{a["url"]}">{a["name"]}</a></p>'
        try:
            db.execute("INSERT INTO gov_raw (title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                       (art["title"], body, art["url"], art["pub_date"], SITE_NAME))
            db.commit()
            saved += 1
        except Exception as e:
            log(f"DB写入失败: {art['title'][:30]} {e}")
        if saved % 100 == 0 and saved > 0:
            log(f"进度: {saved} 存, {skipped} 跳")
        time.sleep(0.3)
    db.close()
    log(f"完成! 新增 {saved} 条, 跳过 {skipped} 条")

if __name__ == "__main__":
    main()
