#!/usr/bin/env python3
"""
兴宁市生态环境局-建设项目环境影响评价信息(jsxmhjyxpjxx)
URL: http://www.xingning.gov.cn/zfjg/xnshjbhj/hbxx/jsxmhjyxpjxx/index.html
分页: index.html / index_2.html ... index_6.html
约66条, NFCMS(南方网)系统
"""

import requests
import sqlite3
import json
import os
import argparse
from bs4 import BeautifulSoup

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "兴宁市-环评信息"
BASE_URL = "https://www.xingning.gov.cn"
LIST_PATH = "/zfjg/xnshjbhj/hbxx/jsxmhjyxpjxx"
TOTAL_PAGES = 6

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def get_soup(url, timeout=20):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = "utf-8"
    return BeautifulSoup(r.text, "html.parser")


def parse_list(soup):
    """提取列表页的文章链接"""
    items = []
    for a in soup.select("a[href*='content/post_']"):
        href = a.get("href", "")
        if not href.startswith("http"):
            href = BASE_URL + href
        title = a.get_text(strip=True)
        items.append((href, title))
    return items


def extract_detail(soup, url):
    """提取详情页内容"""
    title = ""
    pub_date = ""
    content = ""
    attachments = []

    # Title
    nt = soup.select_one(".is-newstitle")
    if nt:
        title = nt.get_text(strip=True)

    # Meta date
    meta_pd = soup.find("meta", attrs={"name": "PubDate"})
    if meta_pd and meta_pd.get("content"):
        pub_date = meta_pd["content"].strip()

    # Content - data table
    content_div = soup.select_one(".is-newscontnet")
    if content_div:
        # 迭代子元素：表格保留HTML，其他取文本
        text_parts = []
        for child in content_div.children:
            if child.name == "table":
                text_parts.append(str(child))
                text_parts.append("")
            elif child.name in ("p", "div", "span"):
                txt = child.get_text(strip=True)
                if txt:
                    text_parts.append(txt)
            elif isinstance(child, str):
                txt = child.strip()
                if txt:
                    text_parts.append(txt)
        content = "\n".join(text_parts)

        # Attachments
        for a_tag in content_div.find_all("a"):
            a_href = a_tag.get("href", "")
            if a_href.endswith((".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar")):
                if not a_href.startswith("http"):
                    a_href = BASE_URL + a_href
                attachments.append({
                    "name": a_tag.get_text(strip=True) or os.path.basename(a_href),
                    "url": a_href,
                })

    return title, pub_date, content, attachments


def crawl(max_pages=None):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()

    pages = max_pages if max_pages else TOTAL_PAGES
    all_items = []

    for page in range(1, pages + 1):
        if page == 1:
            url = f"{BASE_URL}{LIST_PATH}/index.html"
        else:
            url = f"{BASE_URL}{LIST_PATH}/index_{page}.html"

        print(f"[列表] 第{page}/{pages}页: {url}")
        try:
            soup = get_soup(url)
            items = parse_list(soup)
            print(f"  -> {len(items)} 条")
            all_items.extend(items)
        except Exception as e:
            print(f"  -> 失败: {e}")

    new_count = 0
    error_count = 0

    for idx, (page_url, list_title) in enumerate(all_items, 1):
        cur.execute(
            "SELECT id FROM gov_raw WHERE page_url=? AND site_name=?",
            (page_url, SITE_NAME),
        )
        if cur.fetchone():
            print(f"  [{idx}/{len(all_items)}] 跳过: {list_title[:30]}")
            continue

        print(f"  [{idx}/{len(all_items)}] {list_title[:40]}...")
        try:
            soup = get_soup(page_url)
            title, pub_date, content, attachments = extract_detail(soup, page_url)
            if not title:
                title = list_title

            att_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""

            cur.execute(
                """INSERT OR REPLACE INTO gov_raw
                (page_url, site_name, title, publish_date, source_url, content, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?)""",
                (page_url, SITE_NAME, title, pub_date, page_url, content, att_json),
            )
            conn.commit()
            new_count += 1
            print(f"    -> 新增")
        except Exception as e:
            error_count += 1
            print(f"    -> 异常: {e}")

    conn.close()
    return new_count, len(all_items), error_count


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None)
    args = parser.parse_args()

    new, total, errors = crawl(args.max_pages)
    print(f"\n完成: 新增 {new}, 列表 {total}, 异常 {errors}")
