#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""爬取英德市 - 建设项目环境影响评价信息 (修复版：处理乱序列表)"""
import requests, sqlite3, re, sys, time
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE = "http://www.yingde.gov.cn/zljs/zdlyxxgk/hjbh/jsxmhjyxpjxx"
DB = "/root/search.db"
CUTOFF = datetime.now() - timedelta(days=365 * 3)
SITE_NAME = "英德市-建设项目环评信息"

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
HEADERS = {"User-Agent": UA, "Accept-Language": "zh-CN,zh;q=0.9"}

s = requests.Session()
s.headers.update(HEADERS)


def parse_list(html):
    """解析列表页所有条目（不按日期中断）"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.select_one("div.list-right > ul")
    if not ul:
        return items
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a or not a.get("href"):
            continue
        href = urljoin(BASE, a["href"])
        title = a.get_text(strip=True)
        if not title:
            continue
        span = li.find("span")
        date_str = span.get_text(strip=True) if span else ""
        items.append((title, href, date_str))
    return items


def fetch_detail(url):
    """抓取详情页，返回(title, content)"""
    try:
        r = s.get(url, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        return None, None
    
    soup = BeautifulSoup(r.text, "html.parser")
    
    # 标题
    title_tag = soup.find("title")
    if not title_tag:
        return None, None
    title = title_tag.get_text(strip=True)
    # 去除标题前后缀
    for suffix in ["_英德市人民政府网", "_英德市人民政府", " - 英德市人民政府", " - 英德市"]:
        if title.endswith(suffix):
            title = title[:-len(suffix)].strip()
    prefix = "建设项目环境影响评价信息_"
    if title.startswith(prefix):
        title = title[len(prefix):]
    
    # 正文 - 寻找内容区域
    content = ""
    
    # 策略1: 查找页面中所有表格（英德市特色，内容在table里）
    tables = soup.find_all("table")
    if tables:
        parts = []
        for table in tables:
            txt = table.get_text(strip=True)
            if len(txt) > 20:
                rows = []
                for tr in table.find_all("tr"):
                    cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                    if any(cells):
                        rows.append(" | ".join(cells))
                if rows:
                    parts.append("\n".join(rows))
        if parts:
            content = "\n\n".join(parts)
    
    # 策略2: 定位shareIcon后面的内容区
    if not content:
        share = soup.select_one(".content_shareIcon")
        if share:
            for elem in share.find_next_siblings():
                txt = elem.get_text(strip=True)
                if len(txt) > 20 and "分享" not in txt and "二维码" not in txt:
                    parts = []
                    for p in elem.find_all("p"):
                        pt = p.get_text(strip=True)
                        if pt and len(pt) > 5:
                            parts.append(pt)
                    content = "\n\n".join(parts)
                    if content:
                        break
    
    # 策略3: 尝试常见选择器
    if not content:
        for selector in [".news-content", ".content", "#content", ".article-content", ".main-content"]:
            div = soup.select_one(selector)
            if div:
                parts = []
                for p in div.find_all("p"):
                    txt = p.get_text(strip=True)
                    if txt and len(txt) > 5:
                        parts.append(txt)
                for table in div.find_all("table"):
                    rows = []
                    for tr in table.find_all("tr"):
                        cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                        if any(cells):
                            rows.append(" | ".join(cells))
                    if rows:
                        parts.append("\n".join(rows))
                content = "\n\n".join(parts)
                if content:
                    break
    
    # 策略4: 回退：所有p标签
    if not content:
        parts = []
        for p in soup.find_all("p"):
            txt = p.get_text(strip=True)
            if len(txt) > 10:
                parts.append(txt)
        content = "\n\n".join(parts)
    
    return title, content


def main():
    print("[英德市-建设项目环评信息] 开始爬取", flush=True)
    
    url = BASE + "/index.html"
    r = s.get(url, timeout=30)
    r.encoding = "utf-8"
    
    items = parse_list(r.text)
    print("  列表页共 {} 条".format(len(items)), flush=True)
    
    conn = sqlite3.connect(DB)
    c = conn.cursor()
    
    new_count = 0
    for title, page_url, date_str in items:
        # 日期过滤
        try:
            pub_date = datetime.strptime(date_str, "%Y-%m-%d")
            if pub_date < CUTOFF:
                print("  [SKIP] 超3年: {} - {}".format(date_str, title[:40]), flush=True)
                continue
        except:
            pass
        
        # 查重
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,))
        if c.fetchone():
            print("  [EXISTS] {}".format(title[:40]), flush=True)
            continue
        
        print("  [FETCH] {}".format(title[:40]), flush=True)
        detail_title, content = fetch_detail(page_url)
        if not detail_title or not content:
            print("  [SKIP] 无内容: {}".format(title[:40]), flush=True)
            continue
        
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, content, summary, publish_date, site_name) VALUES (?,?,?,?,?,?)",
                (detail_title, page_url, content, content[:500], date_str, SITE_NAME)
            )
            if c.rowcount > 0:
                conn.commit()
                new_count += 1
                print("  [SAVED] {} | len={}".format(detail_title[:40], len(content)), flush=True)
        except Exception as e:
            print("  [DB ERROR] {}".format(e), flush=True)
            conn.rollback()
        
        time.sleep(0.3)
    
    conn.close()
    print("完成! 新增 {} 条".format(new_count), flush=True)


if __name__ == "__main__":
    main()
