#!/usr/bin/env python3
"""中新苏滁高新技术产业开发区 - 公示公告 (仅第1页，AJAX被屏蔽)"""
import os, sys, re, requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
import sqlite3

BASE_URL = "https://scp.chuzhou.gov.cn"
LIST_URL = BASE_URL + "/zwgk/tzgg/gsgg/index.html"
SITE_NAME = "中新苏滁高新区-公示公告"
DATE_THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
import urllib3
urllib3.disable_warnings()

def parse_list(html):
    items = []
    for m in re.finditer(
        r'<a href="(https://scp\.chuzhou\.gov\.cn[^"]+)"[^>]*title="([^"]*)"[^>]*class="left"[^>]*>.*?</a>.*?<span class="right date">([^<]+)</span>',
        html, re.DOTALL
    ):
        items.append({"title": m.group(2), "url": m.group(1), "date": m.group(3).strip()})
    return items

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        soup = BeautifulSoup(r.text, "html.parser")
        content = ""
        cd = soup.select_one("div.j-fontContent.newscontnet")
        if cd:
            content = str(cd)
        source = ""
        ms = soup.find("meta", attrs={"name": "ContentSource"})
        if ms:
            source = ms.get("content", "")
        return content, source
    except Exception as e:
        return "", ""

def main():
    try:
        r = requests.get(LIST_URL, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        items = parse_list(r.text)
    except Exception as e:
        print(f"[ERROR] list page: {e}")
        return

    print(f"共{len(items)}条")

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    c = conn.cursor()
    existing = set(
        row[0] for row in c.execute(
            "SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall()
    )

    new_count = 0
    skip_count = 0
    for item in items:
        if item["date"] < DATE_THRESHOLD:
            skip_count += 1
            continue
        print(f"  {item['date']} {item['title'][:50]}...")
        if item["url"].strip() in existing:
            skip_count += 1
            print(f"    - 已存在")
            continue
        content, source = fetch_detail(item["url"])
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (title, page_url, publish_date, content, site_name)
            VALUES (?, ?, ?, ?, ?)
        """, (item["title"].strip(), item["url"].strip(), item["date"], content, SITE_NAME))
        if c.rowcount > 0:
            new_count += 1
            print(f"    ✓ 新增")
        else:
            skip_count += 1
            print(f"    - 已存在")
        conn.commit()

    conn.close()
    print(f"\n=== 完成 ===")
    print(f"新增: {new_count}, 跳过: {skip_count}")

if __name__ == "__main__":
    main()
