#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""辽宁（营口）沿海产业基地 — 通知公告爬虫（EpointWebBuilder REST API）"""
import requests, sqlite3, re, json
from bs4 import BeautifulSoup

DB = "/root/search.db"
BASE = "https://ykcyjd.yingkou.gov.cn"
API_URL = BASE + "/EWB_YK_Mid/rest/lightfrontaction/getpageinfolist"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
           "Content-Type": "application/json",
           "Referer": BASE + "/003/003003/about.html"}
SITE = "ykcyjd_tzgg"
MAX_PAGES = 40
SITE_GUID = "f404f469-5cb5-4592-8007-5ca1add3b814"
CATEGORY_NUM = "003003"
seen_urls = set()

def fetch_list(page_index):
    payload = {"token": "", "params": {
        "siteGuid": SITE_GUID,
        "categoryNum": CATEGORY_NUM,
        "pageIndex": page_index
    }}
    try:
        r = requests.post(API_URL, json=payload, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        data = r.json()
        if data.get("status", {}).get("code") != 100:
            return [], 0
        custom = data.get("custom", {})
        if isinstance(custom, str):
            custom = json.loads(custom)
        items = custom.get("data", [])
        total = data.get("controls", [{}])[0].get("total", 0) if data.get("controls") else 0
        return items, total
    except Exception as e:
        print(f"  [ERR] API请求失败: {e}")
        return [], 0

def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] 详情请求失败: {e}")
        return None, None, None
    soup = BeautifulSoup(r.text, "html.parser")
    # 标题
    title = ""
    tt = soup.select_one("div.ewb-article-tt h3")
    if tt:
        title = tt.get_text(strip=True)
    # 日期
    date = ""
    src_div = soup.find("div", class_="ewb-source")
    if src_div:
        for p in src_div.find_all("p"):
            txt = p.get_text(strip=True)
            m = re.search(r"(\d{4}-\d{2}-\d{2})", txt)
            if m:
                date = m.group(1)
                break
    # 正文
    content_div = soup.find("div", class_="ewb-article-content", id="ivs_content")
    content = ""
    if content_div:
        for tag in content_div.find_all(["p", "div", "table"]):
            txt = str(tag).strip()
            if txt:
                content += txt + "\n"
        content = content.strip()
    return title, date, content

def main():
    conn = sqlite3.connect(DB)
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()
    c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search_v3 USING fts5(title, content, source_url, publish_date, site_name, tokenize='trigram')")
    
    total_new = total_dup = total_skip = 0
    total_records = 0
    
    for pg in range(0, MAX_PAGES):
        print(f"--- 第{pg+1}页 (pageIndex={pg}) ---")
        items, total = fetch_list(pg)
        if not items:
            print("  无结果，停止翻页")
            break
        if pg == 0:
            total_records = total
            total_pages = (total + 14) // 15
            print(f"  总计{total}条, {total_pages}页")
        
        print(f"  找到{len(items)}条")
        for item in items:
            url = item.get("infourl", "")
            if not url.startswith("http"):
                url = BASE + url
            title = item.get("realtitle", "") or item.get("title", "")
            date = item.get("infodate", "")
            
            if url in seen_urls:
                continue
            seen_urls.add(url)
            
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                total_dup += 1
                continue
            print(f"  [{date}] {title[:60]}")
            dtitle, ddate, content = extract_detail(url)
            if not dtitle: dtitle = title
            if not ddate: ddate = date
            if not content or len(content) < 200:
                print(f"    正文过短({len(content) if content else 0}), 跳过")
                total_skip += 1
                continue
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            try:
                c.execute("INSERT INTO gov_raw (title, summary, content, page_url, publish_date, site_name) VALUES (?,?,?,?,?,?)",
                    (dtitle, summary, content, url, ddate, SITE))
                c.execute("INSERT INTO gov_search_v3 (title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                    (dtitle, content, url, ddate, SITE))
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_dup += 1
    
    print(f"\n新增: {total_new}  重复: {total_dup}  过短: {total_skip}  总计: {total_new+total_dup+total_skip}")

if __name__ == "__main__":
    main()
