#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""岳阳市人民政府 — 生态环境信息公开爬虫 (TRS CMS)"""
import requests, sqlite3, re, sys, os, argparse, subprocess, urllib.parse
from bs4 import BeautifulSoup

DB = "/root/search.db"
BASE = "http://www.yueyang.gov.cn"
LIST_BASE = BASE + "/web/2570/2587/5862"
SITE = "岳阳市-生态环境"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
MAX_PAGES = 5

seen_urls = set()

def decode_gb2312(content):
    """将gb2312内容转为utf-8"""
    if isinstance(content, bytes):
        return content.decode("gb2312", errors="replace")
    return content

def extract_detail(url):
    """提取详情页标题、日期、内容"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "gb2312"
        html = r.text
    except Exception as e:
        print(f"  [ERR] 详情请求失败: {e}")
        return None, None, None
    soup = BeautifulSoup(html, "html.parser")
    # 标题 - ArticleTitle meta
    title = ""
    meta_t = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_t and meta_t.get("content"):
        title = meta_t["content"].strip()
    if not title:
        t_tag = soup.find("title")
        if t_tag:
            title = t_tag.get_text(strip=True)
            title = re.sub(r"\s*[—\-–|]\s*.*", "", title).strip()
    # 日期
    date = ""
    meta_pub = soup.find("meta", attrs={"name": "PubDate"})
    if meta_pub and meta_pub.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_pub["content"])
        if m:
            date = m.group(1)
    # 正文
    content = ""
    zoom = soup.find("div", id="zoom")
    if zoom:
        content = str(zoom).strip()
    if not content or len(content) < 20:
        content_div = soup.find("div", class_="content")
        if content_div:
            content = str(content_div).strip()
    return title, date, content


def parse_list(html, current_url):
    """解析列表页"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for tr in soup.select("div.treeTable table tr"):
        # 跳过表头行
        if tr.find("th"):
            continue
        tds = tr.find_all("td")
        if len(tds) < 3:
            continue
        date_td = tds[2]
        title_td = tds[1]
        a = title_td.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        # 相对路径转换
        href = urllib.parse.urljoin(current_url, href)
        title = a.get("title", "") or a.get_text(strip=True)
        title = title.strip()
        if not title or href in seen_urls:
            continue
        seen_urls.add(href)
        date = date_td.get_text(strip=True)
        items.append({"title": title, "url": href, "date": date})
    return items


def main():
    parser = argparse.ArgumentParser(description="岳阳市-生态环境信息爬虫")
    parser.add_argument("--pages", type=int, default=MAX_PAGES, help="最大爬取页数")
    args = parser.parse_args()

    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    total_new = total_dup = 0

    for pg in range(1, args.pages + 1):
        if pg == 1:
            url = LIST_BASE + "/default.htm"
        else:
            url = LIST_BASE + f"/default_{pg-1}.htm"
        print(f"[PAGE {pg}] {url}")
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = "gb2312"
            html = r.text
        except Exception as e:
            print(f"  [ERR] 列表请求失败: {e}")
            break

        if "暂无数据" in html or "没有记录" in html:
            print("  [END] 无更多数据")
            break

        items = parse_list(html, url)
        if not items:
            print("  [END] 无更多条目")
            break

        print(f"  [LIST] 本页 {len(items)} 条")

        for item in items:
            title = item["title"]
            url = item["url"]
            date = item["date"]

            c.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
            if c.fetchone():
                print(f"  [DUP] {title[:40]} - {date}")
                total_dup += 1
                continue

            dt, dd, dc = extract_detail(url)
            det_title = dt or title
            det_date = dd or date
            content = dc or ""

            if len(content) < 20:
                print(f"  [WARN] 正文过短 ({len(content)}字符): {det_title[:30]}")

            try:
                c.execute(
                    "INSERT INTO gov_raw (title, content, publish_date, source_url, page_url, site_name, group_name, industry) VALUES (?,?,?,?,?,?,?,?)",
                    (det_title, content, det_date, BASE, url, SITE, "湖南", "政府公告"),
                )
                conn.commit()
                print(f"  [NEW] {det_title[:40]} - {det_date}")
                total_new += 1
            except sqlite3.IntegrityError as e:
                print(f"  [DUP] {det_title[:40]} - {e}")
                total_dup += 1

    # FTS
    print(f"\n{'='*50}")
    print(f"新增: {total_new} | 重复: {total_dup}")
    if total_new > 0:
        print("同步FTS...")
        rows = c.execute(
            "SELECT id, title, site_name, content FROM gov_raw WHERE id NOT IN (SELECT rowid FROM gov_search)"
        ).fetchall()
        if rows:
            sql = "INSERT INTO gov_search(rowid, title, site_name, summary) VALUES\n"
            vals = []
            for row_id, t, sn, _ in rows:
                summary = (_ or "")[:500].replace("'", "''")
                t_esc = (t or "").replace("'", "''")
                sn_esc = (sn or "").replace("'", "''")
                vals.append(f"({row_id},'{t_esc}','{sn_esc}','{summary}')")
            sql += ",\n".join(vals) + ";"
            proc = subprocess.run(
                ["sqlite3", "-cmd", ".timeout 60000", DB], input=sql, capture_output=True, text=True, timeout=30
            )
            if proc.returncode == 0:
                print(f"FTS同步完成: {len(rows)}条")
            else:
                print(f"FTS同步失败: {proc.stderr}")
        else:
            print("FTS无需同步")
    conn.close()
    print(f"完成! 共 {total_new} 新, {total_dup} 重复")


if __name__ == "__main__":
    main()
