#!/usr/bin/env python3
"""石棉县人民政府 — 意见征集爬虫
http://www.shimian.gov.cn/xinwen/list/d89a426c-531d-4595-a58d-10bf11e3918c.html
CMS自定义, ?page=N 分页, 20条/页, 14页~273条
内容含正文+PDF附件
"""
import os, sys, re, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.shimian.gov.cn/xinwen/list/d89a426c-531d-4595-a58d-10bf11e3918c.html"
SITE_NAME = "石棉县人民政府-意见征集"
THRESHOLD = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
MAX_PAGES = 14

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9;q=0.8",
    "Referer": "http://www.shimian.gov.cn/",
}
session = requests.Session()
session.headers.update(HEADERS)
import sqlite3

def store_item(title, date_str, content, page_url):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=10)
        c = conn.cursor()
        summary = (content.strip()[:200] if content.strip() else '')
        date_rank = 0
        if date_str:
            try:
                date_rank = int(datetime.strptime(date_str[:10], "%Y-%m-%d").timestamp())
            except:
                pass
        c.execute("""INSERT OR IGNORE INTO gov_raw (title, publish_date, content, page_url, source_url, site_name, summary, date_rank)
                     VALUES (?,?,?,?,?,?,?,?)""",
                  (title.strip(), date_str, content.strip(), page_url.strip(), page_url.strip(), SITE_NAME, summary, date_rank))
        a = c.rowcount; conn.commit(); conn.close(); return a
    except Exception as e:
        print(f"[ERROR] 写入失败: {e}"); return 0

def parse_list_page(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    pl = soup.find(class_="page_list")
    if not pl: return items
    for li in pl.find_all("li"):
        a = li.find("a")
        if not a: continue
        href, text = a.get("href", ""), a.get_text(strip=True)
        if not text or len(text) < 5: continue
        span = li.find("span")
        date_str = span.get_text(strip=True) if span else ""
        full_url = href if href.startswith("http") else urljoin("http://www.shimian.gov.cn", href)
        items.append((text, date_str, full_url))
    return items

def parse_detail_page(html, page_url):
    soup = BeautifulSoup(html, "html.parser")

    # Title
    title = ""
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"): title = mt["content"]
    if not title:
        h1 = soup.select_one("div.msg-content h1")
        if h1: title = h1.get_text(strip=True)
    if not title:
        tt = soup.find("title")
        if tt: title = tt.get_text(strip=True).replace(" - 石棉县人民政府", "")

    # Date
    date_str = ""
    mp = soup.find("meta", attrs={"name": "PubDate"})
    if mp and mp.get("content"):
        m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", mp["content"])
        if m: date_str = m.group(1)
    if not date_str:
        intro = soup.find(class_="intro")
        if intro:
            for span in intro.find_all("span"):
                m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", span.get_text())
                if m: date_str = m.group(1); break

    # 1) Extract body text
    body_parts = []
    xqb = soup.find(class_="xqing-web-box")
    if xqb:
        for p in xqb.find_all("p"):
            t = p.get_text(strip=True)
            if t and len(t) > 3: body_parts.append(t)
        for table in xqb.find_all("table"):
            rows = []
            for tr in table.find_all("tr"):
                cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                if cells: rows.append(" | ".join(cells))
            if rows: body_parts.append("表格:\n" + "\n".join(rows))
    else:
        ib = soup.find(class_="id_body")
        if ib:
            for elem in ib.children:
                if elem.name != "div": continue
                style = elem.get("style", "")
                text = elem.get_text(strip=True)
                if "font-size" in style or "bold" in style: continue
                if re.match(r'^(发布时间|来源|作者|浏览次数|字体)', text): continue
                if text and len(text) > 3: body_parts.append(text)
        if not body_parts:
            for p in soup.find_all("p"):
                t = p.get_text(strip=True)
                if t and len(t) > 5: body_parts.append(t)

    seen = set()
    unique = []
    for part in body_parts:
        if part not in seen: seen.add(part); unique.append(part)
    content = "\n\n".join(unique)

    # 2) Attachments
    atts = []
    view = soup.find(class_="view")
    if view:
        for iframe in view.find_all("iframe"):
            src = iframe.get("src", "")
            if src: atts.append("[PDF原文](" + src + ")")
        for tag in view.find_all(["object", "embed"]):
            src = tag.get("data") or tag.get("src", "")
            if src: atts.append("[附件](" + src + ")")
    files = soup.find(class_="files")
    if files:
        for a in files.find_all("a", href=True):
            t = a.get_text(strip=True) or "附件"
            h = a["href"]
            if not h.startswith("http"):
                h = "http://www.shimian.gov.cn" + h
            atts.append("[" + t + "](" + h + ")")
    for a in soup.find_all("a", href=True):
        h = a["href"]
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|caj)$", h, re.I):
            t = a.get_text(strip=True) or "附件"
            fu = h if h.startswith("http") else "http://www.shimian.gov.cn" + h
            link = "[" + t + "](" + fu + ")"
            if link not in atts: atts.append(link)

    if atts:
        if content:
            content += "\n\n---\n附件：\n" + "\n".join(atts)
        else:
            content = "附件：\n" + "\n".join(atts)

    return title, date_str, content

def main():
    print(f"[INFO] 开始爬取: {SITE_NAME}")
    print(f"[INFO] 阈值: {THRESHOLD}, 最大页数: {MAX_PAGES}")
    total_new, stop = 0, False
    for page_num in range(1, MAX_PAGES + 1):
        if stop: break
        page_url = BASE_URL if page_num == 1 else f"{BASE_URL}?page={page_num}"
        if page_num > 1:
            time.sleep(1.5)
        try:
            r = session.get(page_url, timeout=15)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"[ERROR] 请求失败: {e}"); time.sleep(1); continue
        if r.status_code != 200: break
        items = parse_list_page(r.text)
        if not items: break
        print(f"[PAGE] 第{page_num}页: {len(items)}条 [{items[0][1]}] {items[0][0][:35]}")
        for title, date_str, detail_url in items:
            if date_str and date_str < THRESHOLD:
                print(f"[STOP] {date_str} 超阈值"); stop = True; break
            time.sleep(1.5)
            try:
                rd = session.get(detail_url, timeout=15)
                rd.encoding = "utf-8"
            except: continue
            if rd.status_code != 200: continue
            dt, dd, dc = parse_detail_page(rd.text, detail_url)
            ft = dt if dt else title
            fd = dd or date_str
            affected = store_item(ft, fd, dc, detail_url)
            if affected > 0:
                total_new += 1
                print(f"  [+] {ft[:35]}...({fd})")
            else:
                print(f"  [-] 已存在")
    print(f"\n=== 完成, 新增: {total_new} ===")

if __name__ == "__main__":
    main()
