#!/usr/bin/env python3
"""续爬 jinjiang.gov.cn 剩余页面"""
import json, requests, time, re
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
import sqlite3

requests.packages.urllib3.disable_warnings()

SITE_NAME = "jinjiang.gov.cn-环评信息"
API_URL = "https://www.jinjiang.gov.cn/ssp/search/api/v2/external"
API_H = {"User-Agent": "Mozilla/5.0", "Content-Type": "application/json"}
DETAIL_H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
CUTOFF_TS = int(datetime.strptime(CUTOFF, "%Y-%m-%d").timestamp() * 1000)

session = requests.Session()
session.headers.update(DETAIL_H)

def fetch_detail(url):
    try:
        r = session.get(url, timeout=15)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None
        soup = BeautifulSoup(r.text, "html.parser")
        trs = soup.select_one(".TRS_Editor")
        if trs:
            return str(trs)
        area = soup.select_one(".article_area")
        if area:
            font = area.select_one("div[class*=font]")
            return str(font) if font else str(area)
        return None
    except Exception as e:
        print(f"  [ERR] detail: {e}")
        return None

conn = sqlite3.connect("/root/search.db")
conn.execute("PRAGMA busy_timeout=15000")
cur = conn.cursor()

total_new = 0
for page in range(1, 50):
    body = {"siteId":"000000008fd72c27018fdd69fb8b0002","apiName":"WCMOPENINFOAPI",
            "pageSize":100,"page":page,"sortField":"docreltime desc",
            "filter":{"docstatus":[{"eq":10}],"chnlid":[{"eq":"38080"}]}}
    try:
        r = requests.post(API_URL, json=body, headers=API_H, verify=False, timeout=20)
        res = r.json()
        result = res.get("data",{}).get("result",[])
        if not result:
            break
    except Exception as e:
        print(f"Page {page} API error: {e}")
        break

    all_old = True
    page_new = 0
    for item in result:
        pubdate_ts = int(item.get("pubdate", 0))
        if pubdate_ts < CUTOFF_TS:
            continue
        all_old = False

        url = item.get("docpuburl", "")
        title = item.get("doctitle", "")
        if not url or not title:
            continue

        cur.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url=?", (url,))
        if cur.fetchone()[0] > 0:
            continue

        date_str = datetime.fromtimestamp(pubdate_ts/1000).strftime("%Y-%m-%d")
        content = fetch_detail(url)
        if not content:
            print(f"  [WARN] fallback to text for {title[:40]}")
            content = item.get("doccontent") or item.get("_doccontent") or ""

        date_rank = int(date_str.replace("-","")) if date_str else 0
        try:
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name,title,page_url,publish_date,content,date_rank,category) VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, title, url, date_str, content, date_rank, "hjxx")
            )
            if cur.rowcount > 0:
                page_new += 1
                total_new += 1
                if page_new % 10 == 0:
                    conn.commit()
        except Exception as e:
            print(f"  [ERR] insert: {e}")
        time.sleep(0.3)

    conn.commit()
    print(f"Page {page}: +{page_new} (total +{total_new})", flush=True)

    if all_old and int(result[-1].get("pubdate", 0)) < CUTOFF_TS:
        print("All remaining before cutoff, stopping.")
        break

conn.close()
print(f"Done. Total new: {total_new}")
