#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""邹城市人民政府 — 项目环评和验收爬虫"""
import requests, sqlite3, re
from bs4 import BeautifulSoup

DB = "/root/search.db"
BASE = "http://www.zoucheng.gov.cn"
# Hanweb dataproxy API - much faster than scraping list pages
DATAPROXY = BASE + "/module/web/jpage/dataproxy.jsp?page={}&webid=103&path=/&columnid=37863&unitid=455187&permissiontype=0"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36"}
SITE = "zoucheng_hjhp"
MAX_PAGES = 20
seen_urls = set()

def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] 详情请求失败: {e}")
        return None, None, None
    soup = BeautifulSoup(r.text, "html.parser")
    # 标题
    title = ""
    tt = soup.select_one("div.main_tit")
    if tt:
        title = tt.get_text(strip=True)
    # 日期
    date = ""
    pub = soup.find("span", id="pubDate")
    if pub:
        d = pub.get_text(strip=True)
        m = re.search(r"(\d{4}-\d{2}-\d{2})", d)
        if m: date = m.group(1)
    if not date:
        bd = soup.find("span", id="buildDate")
        if bd:
            d = bd.get_text(strip=True)
            m = re.search(r"(\d{4}-\d{2}-\d{2})", d)
            if m: date = m.group(1)
    # 正文
    content_div = soup.find("div", class_="neirong")
    content = ""
    if content_div:
        for tag in content_div.find_all(["p", "div", "table"]):
            txt = str(tag).strip()
            if txt: content += txt + "\n"
        content = content.strip()
    return title, date, content

def parse_dataproxy(xml_text):
    """Parse the XML dataproxy response to extract items."""
    items = []
    # Find all CDATA records
    records = re.findall(r'<record><!\[CDATA\[(.*?)\]\]></record>', xml_text, re.DOTALL)
    for rec in records:
        soup = BeautifulSoup(rec, "html.parser")
        a = soup.find("a", href=True)
        if not a: continue
        href = a["href"]
        if not href.startswith("http"):
            href = BASE + href
        title = a.get("title", "") or a.get_text(strip=True)
        if not title or href in seen_urls: continue
        seen_urls.add(href)
        sp = soup.find("span", class_="bt-right")
        date = sp.get_text(strip=True) if sp else ""
        items.append({"title": title.strip(), "url": href, "date": date})
    return items

def main():
    conn = sqlite3.connect(DB)
    conn.execute("PRAGMA journal_mode=WAL")
    c = conn.cursor()
    c.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search_v3 USING fts5(title, content, source_url, publish_date, site_name, tokenize='trigram')")
    
    total_new = total_dup = total_skip = 0
    
    for pg in range(1, MAX_PAGES + 1):
        url = DATAPROXY.format(pg)
        print(f"--- 第{pg}页 ---")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
        except Exception as e:
            print(f"  [ERR] dataproxy请求失败: {e}")
            break
        
        # Check for totalpage
        tp = re.search(r'<totalpage>(\d+)</totalpage>', r.text)
        if tp:
            total_pages = int(tp.group(1))
            if total_pages < pg:
                print(f"  已到最后一页({total_pages}页)，停止")
                break
            if pg == 1:
                tr = re.search(r'<totalrecord>(\d+)</totalrecord>', r.text)
                total_rec = tr.group(1) if tr else "?"
                print(f"  总计{total_pages}页, {total_rec}条")
        
        items = parse_dataproxy(r.text)
        if not items:
            print("  无结果，停止翻页")
            break
        print(f"  找到{len(items)}条")
        
        for item in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if c.fetchone():
                total_dup += 1
                continue
            print(f"  [{item['date']}] {item['title'][:60]}")
            title, date, content = extract_detail(item["url"])
            if not title: title = item["title"]
            if not date: date = item["date"]
            if not content or len(content) < 200:
                print(f"    正文过短({len(content) if content else 0}), 跳过")
                total_skip += 1
                continue
            summary = re.sub(r"<[^>]+>", "", content)[:200]
            summary = re.sub(r"\s+", " ", summary).strip()
            try:
                c.execute("INSERT INTO gov_raw (title, summary, content, page_url, publish_date, site_name) VALUES (?,?,?,?,?,?)",
                    (title, summary, content, item["url"], date, SITE))
                c.execute("INSERT INTO gov_search_v3 (title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                    (title, content, item["url"], date, SITE))
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_dup += 1
    
    print(f"\n新增: {total_new}  重复: {total_dup}  过短: {total_skip}  总计: {total_new+total_dup+total_skip}")

if __name__ == "__main__":
    main()
