#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""广汉市人民政府 - 动态信息爬虫"""
import requests, sqlite3, re
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE_URL = "https://www.guanghan.gov.cn"
LIST_URL = "https://www.guanghan.gov.cn/info/iList.jsp?cat_id=24540&cur_page={}"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36", "Referer": "https://www.guanghan.gov.cn/"}
SITE_NAME = "广汉市-通知公告"
MAX_PAGES = 100
seen_urls = set()

def _drop_container_parts(parts):
    """剔除「容器段」：find_all(['p','div']) 会同时收下容器 <div> 与它内部的 <p>，
    导致同一内容重复（容器那份常还被 get_text(strip=True) 拍平）。
    判据：先按值去重，再剔除被其它段完全包含的段。保护：剔除后为空则返回去重结果。
    """
    if not parts:
        return parts
    ps = [p for p in parts if isinstance(p, str)]
    if len(ps) != len(parts):
        return parts
    seen, uniq = set(), []
    for p in parts:
        if p not in seen:
            seen.add(p); uniq.append(p)
    keep = [a for a in uniq
            if not (len(a) >= 40 and any(b is not a and b and b in a for b in uniq))]
    return keep if keep else uniq


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn


def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] 详情页请求失败: {e}")
        return None, None, None
    soup = BeautifulSoup(r.text, "html.parser")
    title_tag = soup.find("h2")
    title = title_tag.get_text(strip=True) if title_tag else ""
    date = ""
    attr = soup.find("p", id="attribute")
    if attr:
        for sp in attr.find_all("span"):
            m = re.search(r'(\d{4}-\d{2}-\d{2})', sp.get_text(strip=True))
            if m: date = m.group(1); break
    content_tag = soup.find("article", class_="content") or soup.find("div", class_="content")
    content = ""
    if content_tag:
        _parts = []
        for tag in content_tag.find_all(["p", "div"]):
            txt = str(tag).strip()
            if txt: _parts.append(txt)
        content = "\n".join(_drop_container_parts(_parts))
    return title, date, content

def crawl_page(page_num):
    url = LIST_URL.format(page_num)
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERR] 列表页请求失败: {e}")
        return []
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    for li in soup.select("section ul > li"):
        a_tag = li.find("a", class_="t_ell")
        if not a_tag: continue
        href = a_tag.get("href", "")
        if not href or href == "#": continue
        if not href.startswith("http"): href = BASE_URL + href
        title = a_tag.get("title", "") or a_tag.get_text(strip=True)
        if not title: continue
        if href in seen_urls: continue
        seen_urls.add(href)
        time_tag = li.find("time")
        date = time_tag.get_text(strip=True) if time_tag else ""
        items.append({"title": title, "url": href, "date": date})
    return items

def main():
    conn = get_conn()
    c = conn.cursor()
    total_new = total_dup = total_skip = 0
    for page in range(1, MAX_PAGES + 1):
        print(f"--- 第{page}页 ---")
        items = crawl_page(page)
        if not items:
            print("无结果，停止翻页"); break
        print(f"找到{len(items)}条")
        for item in items:
            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
            if c.fetchone():
                total_dup += 1; continue
            print(f"[{item['date']}] {item['title']}")
            title, date, content = extract_detail(item["url"])
            if not title: title = item["title"]
            if not date: date = item["date"]
            if not content or len(content) < 200:
                print(f"  正文过短({len(content) if content else 0})，跳过")
                total_skip += 1
                continue
            summary = re.sub(r'<[^>]+>', '', content)[:200]
            summary = re.sub(r'\s+', ' ', summary).strip()
            try:
                c.execute("INSERT INTO gov_raw (title, summary, content, page_url, publish_date, site_name) VALUES (?,?,?,?,?,?)",
                    (title, summary, content, item["url"], date, SITE_NAME))
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError as e:
                print(f"  SKIP(integrity): {e}")
                total_dup += 1
    print(f"\n新增: {total_new}  重复: {total_dup}  过短跳过: {total_skip}  总计: {total_new+total_dup+total_skip}")

if __name__ == "__main__":
    main()
