#!/usr/bin/env python3
"""寿阳县人民政府 — 行政审批局通知公告爬虫
http://www.shouyang.gov.cn/zwgk/bmxxgkml/44hzspj/fdzdgknr3/tzgg44hzspj
PowerCMS, tzgg44hzspj_2 ~ tzgg44hzspj_18 分页(10条/页)
"""
import os, sys, re, json, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.shouyang.gov.cn/zwgk/bmxxgkml/44hzspj/fdzdgknr3/tzgg44hzspj"
SITE_NAME = "寿阳县人民政府-行政审批局-通知公告"
THRESHOLD = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
# Page 1 = BASE_URL, page 2 = BASE_URL_2 ... page 18 = last unique
MAX_PAGES = 22

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "http://www.shouyang.gov.cn/",
}
session = requests.Session()
session.headers.update(HEADERS)
import sqlite3

def store_item(title, date_str, content, page_url):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=10)
        c = conn.cursor()
        c.execute("""
            INSERT OR IGNORE INTO gov_raw 
            (title, publish_date, content, page_url, site_name)
            VALUES (?, ?, ?, ?, ?)
        """, (title.strip(), date_str, content.strip(), page_url.strip(), SITE_NAME))
        affected = c.rowcount
        conn.commit()
        conn.close()
        return affected
    except Exception as e:
        print(f"[ERROR] 写入失败: {e}")
        return 0

def parse_list_page(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    info_list = soup.find(class_="infoList")
    if not info_list:
        return items
    for li in info_list.find_all("li"):
        a = li.find("a")
        if not a: continue
        href, text = a.get("href", ""), a.get_text(strip=True)
        if not text or len(text) < 5: continue
        span = li.find("span")
        date_str = span.get_text(strip=True) if span else ""
        full_url = href if href.startswith("http") else "http://www.shouyang.gov.cn" + href if href.startswith("/") else urljoin(BASE_URL + "/", href)
        items.append((text, date_str, full_url))
    return items

def parse_detail_page(html, page_url):
    soup = BeautifulSoup(html, "html.parser")
    h2 = soup.find("h2", class_="title")
    title = h2.get_text(strip=True) if h2 else (soup.title.get_text(strip=True).split("_通知公告")[0] if soup.title else "N/A")
    date_str = ""
    prop = soup.find("div", class_="property")
    if prop:
        m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", prop.get_text())
        if m: date_str = m.group(1)
    content = ""
    con_txt = soup.find(class_="conTxt")
    if con_txt:
        img = con_txt.find("img", attrs={"data-powerurl": True})
        if img:
            pu = img["data-powerurl"]
            pu = ("http://www.shouyang.gov.cn" + pu) if pu.startswith("/") else urljoin(page_url, pu)
            content = f'<p><a href="{pu}">PDF附件</a></p>'
        txt = con_txt.get_text(strip=True)
        if txt and content: content = txt + "\n\n" + content
        elif txt: content = txt
    if not content:
        pa = soup.find(class_="printArea")
        if pa and pa.find("table"):
            rows = [" | ".join(td.get_text(strip=True) for td in tr.find_all("td") if td.get_text(strip=True)) for tr in pa.find("table").find_all("tr")]
            content = "元数据:\n" + "\n".join(rows)
    atts = []
    for a in soup.find_all("a", href=True):
        h = a["href"]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|caj)$', h, re.I):
            t = a.get_text(strip=True) or "附件"
            fu = ("http://www.shouyang.gov.cn" + h) if h.startswith("/") else (h if h.startswith("http") else urljoin(page_url, h))
            atts.append(f"[{t}]({fu})")
    if atts: content += "\n\n---\n附件：\n" + "\n".join(atts)
    return title, date_str, content

def main():
    print(f"[INFO] 开始爬取: {SITE_NAME}")
    print(f"[INFO] 阈值: {THRESHOLD}, 最大页数: {MAX_PAGES}")
    # Page 1: BASE_URL (no suffix). _1 = same as no suffix. _2 = page 2 ...
    urls = [BASE_URL] + [f"{BASE_URL}_{i}" for i in range(2, MAX_PAGES + 1)]
    total_new, stop = 0, False
    for idx, page_url in enumerate(urls):
        if stop: break
        pn = idx + 1
        try:
            r = session.get(page_url, timeout=15)
            r.encoding = "utf-8"
        except:
            time.sleep(1); continue
        if r.status_code != 200: break
        items = parse_list_page(r.text)
        if not items: break
        print(f"[PAGE] 第{pn}页: {len(items)} 条 [{items[0][1]}] {items[0][0][:40]}")
        for title, date_str, detail_url in items:
            if date_str and date_str < THRESHOLD:
                print(f"[STOP] {date_str} 超阈值"); stop = True; break
            time.sleep(0.3)
            try:
                rd = session.get(detail_url, timeout=15)
                rd.encoding = "utf-8"
            except: continue
            if rd.status_code != 200: continue
            dt, dd, dc = parse_detail_page(rd.text, detail_url)
            ft = dt if dt and dt != "N/A" else title
            fd = dd or date_str
            affected = store_item(ft, fd, dc, detail_url)
            if affected > 0:
                total_new += 1
                print(f"  [+] {ft[:40]}... ({fd})")
            else:
                print(f"  [-] 已存在")
    print(f"\n=== 完成, 新增: {total_new} ===")

if __name__ == "__main__":
    main()
