#!/usr/bin/env python3
import os, sys, re, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_LIST = "http://www.shimian.gov.cn/gongkai/jichu/40016.html"
SITE_NAME = "石棉县人民政府-公示公告"
THRESHOLD = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
MAX_PAGES = 120
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36","Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8","Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8","Referer": "http://www.shimian.gov.cn/"}
session = requests.Session()
session.headers.update(HEADERS)
import sqlite3

def store(title, date_str, content, url):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=10)
        c = conn.cursor()
        c.execute("INSERT OR IGNORE INTO gov_raw (title, publish_date, content, page_url, site_name) VALUES (?,?,?,?,?)", (title.strip(), date_str, content.strip(), url.strip(), SITE_NAME))
        a = c.rowcount; conn.commit(); conn.close(); return a
    except: return 0

start_page = int(sys.argv[1]) if len(sys.argv) > 1 else 1
total_new = 0
for page_num in range(start_page, MAX_PAGES + 1):
    page_url = f"{BASE_LIST}?page={page_num}"
    time.sleep(1.5)
    try:
        r = session.get(page_url, timeout=15)
        r.encoding = "utf-8"
    except: time.sleep(2); continue
    if r.status_code != 200: print(f"[PAGE]{page_num}: HTTP {r.status_code}"); break
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    for cn in ["content-list", "data-list"]:
        pl = soup.find(class_=cn)
        if pl: break
    if not pl: print(f"[PAGE]{page_num}: 无数据"); break
    for li in pl.find_all("li"):
        a = li.find("a")
        if not a: continue
        href, text = a.get("href",""), a.get_text(strip=True)
        if not text or len(text) < 5: continue
        span = li.find("span")
        ds = span.get_text(strip=True) if span else ""
        fu = href if href.startswith("http") else urljoin("http://www.shimian.gov.cn", href)
        items.append((text, ds, fu))
    print(f"[PAGE]{page_num}: {len(items)}条 [{items[0][1]}] {items[0][0][:30]}")
    for title, ds, du in items:
        if ds and ds < THRESHOLD: print(f"[STOP]{ds} 超阈值"); sys.exit(0)
        time.sleep(1.5)
        try:
            rd = session.get(du, timeout=15)
            rd.encoding = "utf-8"
        except: continue
        if rd.status_code != 200: continue
        soup2 = BeautifulSoup(rd.text, "html.parser")
        h1 = soup2.select_one("div.msg-content h1")
        dt = h1.get_text(strip=True) if h1 else "N/A"
        dd = ""
        mp = soup2.find("meta", attrs={"name": "PubDate"})
        if mp and mp.get("content"):
            m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", mp["content"])
            if m: dd = m.group(1)
        content = ""
        xqb = soup2.find(class_="xqing-web-box")
        if xqb:
            parts = []
            for p in xqb.find_all("p"):
                t = p.get_text(strip=True)
                if t: parts.append(t)
            for table in xqb.find_all("table"):
                rows = []
                for tr in table.find_all("tr"):
                    cells = [td.get_text(strip=True) for td in tr.find_all(["td","th"])]
                    if cells: rows.append(" | ".join(cells))
                if rows: parts.append("表格:\n" + "\n".join(rows))
            content = "\n\n".join(parts)
        atts = []
        for al in soup2.find_all("a", href=True):
            h = al["href"]
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|caj)$', h, re.I):
                t = al.get_text(strip=True) or "附件"
                fu = h if h.startswith("http") else urljoin("http://www.shimian.gov.cn", h)
                atts.append(f"[{t}]({fu})")
        if atts: content += "\n\n---\n附件：\n" + "\n".join(atts)
        fd = dd or ds
        a = store(dt, fd, content, du)
        if a > 0:
            total_new += 1
            print(f"  [+] {dt[:30]}...({fd})")
print(f"\n=== 完成, 新增: {total_new} ===")
