#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
古蔺县人民政府 - 环境保护 (hjbh2) 爬虫
CMS: 自定义PHP政府信息公开平台
列表: /zwgk/fdzdgknr/zdmsxx/hjbh2 (第1页), /.../hjbh2_N (第N页, N=2~37)
分页: 10条/页, 共365条/37页
详情: /.../hjbh2/content_XXXXX
"""

import re, sys, json, time, requests, sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.gulin.gov.cn"
LIST_FIRST = BASE_URL + "/zwgk/fdzdgknr/zdmsxx/hjbh2"
DB_PATH = "/root/search.db"
SITE_NAME = "古蔺县-环境保护"
CATEGORY = GROUP = "环评"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
MAX_PAGES = 5
TIMEOUT = 30


def init_db():
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn


def fetch(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if attempt == 2:
                print(f"  [WARN] 获取失败 ({attempt+1}/3): {url} - {e}", file=sys.stderr)
                return ""
            time.sleep(2)


def extract_list_items(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for a in soup.select("ul li a"):
        href = a.get("href", "")
        if "content_" not in href:
            continue
        title = a.get_text(strip=True)
        if not title:
            continue
        li = a.find_parent("li")
        date = ""
        if li:
            spans = li.find_all("span")
            if spans:
                date = spans[-1].get_text(strip=True)
        full_url = urljoin(BASE_URL, href)
        items.append((full_url, title, date))
    return items


def extract_detail(html):
    soup = BeautifulSoup(html, "html.parser")

    title_tag = soup.find("title")
    title = ""
    if title_tag:
        raw = title_tag.get_text(strip=True)
        parts = raw.split("_")
        if parts:
            title = parts[0].strip()

    if not title:
        article = soup.find("article")
        if article:
            h2 = article.find("h2")
            if h2:
                title = h2.get_text(strip=True)

    date = ""
    date_match = re.search(r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if date_match:
        date = date_match.group(1)

    content = ""
    attachments = []
    article = soup.find("article")
    if article:
        parts = []
        for p in article.find_all("p"):
            text = p.get_text(strip=True)
            if not text or text in ("【字体：", "小", "大", "】"):
                continue
            for a_tag in p.find_all("a"):
                href = a_tag.get("href", "")
                if href.endswith(".pdf"):
                    fname = a_tag.get_text(strip=True) or href.split("/")[-1]
                    full_url = urljoin(BASE_URL, href)
                    attachments.append({"name": fname, "url": full_url})
            if text and not (len(text) < 5 and any(a_tag.get("href", "").endswith(".pdf") for a_tag in p.find_all("a"))):
                parts.append(text)
        content = "\n\n".join(parts)

    return title, date, content, attachments


def push_to_db(conn, items):
    saved = 0
    skipped = 0
    for url, title, date, content, attachments in items:
        attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
        summary = content[:200].replace("\n", " ") if content else title

        if not content and attachments:
            content = "\n".join(f"[{a['name']}]({a['url']})" for a in attachments)

        try:
            conn.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (page_url, title, site_name, publish_date, content, date_rank, summary, attachments)
                   VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
                (url, title, SITE_NAME, date, content, date, summary, attachments_json)
            )
            if conn.total_changes:
                saved += 1
            else:
                skipped += 1
        except Exception as e:
            print(f"  [ERR] DB写入失败: {url} - {e}", file=sys.stderr)
            skipped += 1
    return saved, skipped


def main():
    pages = MAX_PAGES
    for i, a in enumerate(sys.argv):
        if a == "--pages" and i + 1 < len(sys.argv):
            pages = int(sys.argv[i + 1])
            break

    conn = init_db()
    all_items = []
    list_urls = [LIST_FIRST]
    list_urls.extend(f"{LIST_FIRST}_{n}" for n in range(2, pages + 1))

    print(f"古蔺县-环境保护: 爬取 {pages} 页")

    for idx, list_url in enumerate(list_urls, 1):
        print(f"  [列表页 {idx}/{pages}] {list_url}")
        html = fetch(list_url)
        if not html:
            continue

        items = extract_list_items(html)
        if not items:
            print("    → 无数据，停止")
            break

        print(f"    → 找到 {len(items)} 条")

        for item_url, item_title, item_date in items:
            time.sleep(0.5)
            print(f"    [{len(all_items)+1}] {item_title[:40]}...", end=" ", flush=True)
            detail_html = fetch(item_url)
            if not detail_html:
                print("⚠ 获取失败")
                continue

            title, date, content, attachments = extract_detail(detail_html)
            if not title:
                title = item_title
            if not date:
                date = item_date

            has_table = 1 if "表格" in content or "<table" in detail_html else 0

            all_items.append((item_url, title, date, content, attachments))
            print(f"✅ ({len(content)}字)" if content else "⚠ 空正文")

    if all_items:
        saved, skipped = push_to_db(conn, all_items)
        conn.commit()
        print(f"\n完成! 新增: {saved}, 跳过: {skipped}, 总共: {len(all_items)}")

    conn.close()
    return 0


if __name__ == "__main__":
    sys.exit(main())
