#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
荔波县人民政府 - 环境保护栏目爬虫 (续爬)
从指定页开始，逐页入库
"""
import sys, os, re, requests
from bs4 import BeautifulSoup

sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from crawler_lib import push_to_searchdb, clean_html

BASE_URL = "http://www.libo.gov.cn"
LIST_URL = "http://www.libo.gov.cn/zwgk/xxgkml/zdlyxx/hjbh/"
SITE_NAME = "荔波县人民政府"
INDUSTRY = "环境公示"
START_PAGE = 19
MAX_PAGES = 40
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch_list(page):
    url = f"{BASE_URL}/zwgk/xxgkml/zdlyxx/hjbh/index_{page}.html" if page > 1 else LIST_URL
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        return resp.text
    except:
        return ""

def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for row in soup.select("table.bg_tb tbody#idData tr.c"):
        link_td = row.select_one("td.tn4 a")
        if not link_td: continue
        title = link_td.get_text(strip=True)
        href = link_td.get("href", "").strip()
        if not href.startswith("http"):
            href = BASE_URL + href
        date_td = row.select_one("td.tn5")
        pub_date = date_td.get_text(strip=True) if date_td else ""
        items.append({"title": title, "url": href, "pub_date": pub_date})
    return items

def fetch_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")
        article = soup.select_one("div.article")
        content = clean_html(str(article)) if article else ""
        if not content or len(content) < 50:
            text = soup.get_text(separator="\n")
            lines = [l.strip() for l in text.split("\n") if l.strip() and len(l.strip()) > 10]
            content = "\n".join(lines[10:])
        return content
    except:
        return ""

def main():
    total_new = 0
    for page in range(START_PAGE, MAX_PAGES + 1):
        print(f"[列表] 页{page}...")
        html = fetch_list(page)
        if not html:
            print("  -> 无响应，结束")
            break
        items = parse_list(html)
        if not items:
            print("  -> 无数据，结束")
            break
        print(f"  -> {len(items)} 条")

        batch = []
        for item in items:
            content = fetch_detail(item["url"])
            print(f"  [{'OK' if content else '空'}] {item['title'][:40]} | {item['pub_date']}")
            batch.append({
                "title": item["title"],
                "url": item["url"],
                "source_url": item["url"],
                "site_name": SITE_NAME,
                "pub_date": item["pub_date"],
                "summary": "",
                "content": content,
                "industry": INDUSTRY,
            })

        # 每页立即入库
        if batch:
            try:
                push_to_searchdb(batch, batch_label=f"libo_p{page}")
                total_new += len(batch)
                print(f"  => 入库 {len(batch)} 条")
            except Exception as e:
                print(f"  => 入库错误: {e}")

    print(f"\n===== 完成 =====")
    print(f"新增: {total_new} 条")

if __name__ == "__main__":
    main()
