#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
广德市人民政府 - 建设项目环评文件审批 爬虫
CMS: 自定义PHP政府信息公开平台
列表: /Jczwgk/showList/548/111001001/page_{n}.html
分页: 15条/页, 约200页
详情: /Jczwgk/show/{id}.html
详情标题: <title> (去掉 "--标准化规范化工作专题" 后缀)
详情正文: div.m-dttexts.j-fontContent > p
详情日期: 发布时间：YYYY-MM-DD

用法:
  python3 crawl_guangde_eia.py
  python3 crawl_guangde_eia.py --pages 5
"""

import re
import sys
import time
import requests
import sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.guangde.gov.cn"
LIST_TPL = BASE_URL + "/Jczwgk/showList/548/111001001/page_%d.html"
DB_PATH = "/root/search.db"
SITE_NAME = "广德市-环评审批"
CATEGORY = GROUP = "环评"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
MAX_PAGES = 10
TIMEOUT = 30


def fetch(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            if attempt == 2:
                print("  [WARN] 获取失败 (%d/3): %s - %s" % (attempt + 1, url, e), file=sys.stderr)
                return ""
            time.sleep(2)


def extract_list_items(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.select(".g-listrg .m-liststyle2 ul li"):
        a_tag = li.find("a")
        span_tag = li.find("span")
        if not a_tag:
            continue
        href = a_tag.get("href", "")
        title = a_tag.get("text", "") or a_tag.get("title", "")
        if not href or not title:
            continue
        full_url = urljoin(BASE_URL, href)
        date = span_tag.get_text(strip=True) if span_tag else ""
        items.append((full_url, title, date))
    return items


def extract_detail(html):
    soup = BeautifulSoup(html, "html.parser")

    # 标题：<title> 去掉后缀
    title = ""
    title_tag = soup.find("title")
    if title_tag and title_tag.string:
        title = title_tag.string.strip()
        title = re.sub(r"--标准化规范化工作专题$", "", title).strip()

    # 日期：发布时间：YYYY-MM-DD
    date = ""
    m = re.search(r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", html)
    if m:
        date = m.group(1)

    # 正文：div.m-dttexts > p
    content = ""
    content_div = soup.find("div", class_=lambda c: c and "m-dttexts" in (c if isinstance(c, str) else " ".join(c)))
    if content_div:
        parts = []
        for p in content_div.find_all("p"):
            text = p.get_text(strip=True)
            if text:
                parts.append(text)
        content = "\n\n".join(parts)

    # 附件
    attachments = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if ".pdf" in href.lower() or ".doc" in href.lower() or ".xls" in href.lower():
            name = a.get_text(strip=True) or href.split("/")[-1]
            attachments.append({"name": name, "url": urljoin(BASE_URL, href)})

    return title, date, content, attachments


def push_to_db(conn, items):
    saved = 0
    skipped = 0
    for url, title, date, content, attachments in items:
        import json
        attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
        summary = content[:200].replace("\n", " ") if content else title
        if not content:
            if attachments:
                content = "\n".join("[%s](%s)" % (a["name"], a["url"]) for a in attachments)
            else:
                content = title
        try:
            cur = conn.execute(
                "INSERT OR REPLACE INTO gov_raw "
                "(page_url, title, site_name, publish_date, content, date_rank, summary, attachments) "
                "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                (url, title, SITE_NAME, date, content, date, summary, attachments_json)
            )
            if cur.rowcount > 0:
                saved += 1
            else:
                skipped += 1
        except Exception as e:
            print("  [ERR] DB: %s - %s" % (url, e), file=sys.stderr)
            skipped += 1
    return saved, skipped


def main():
    pages = MAX_PAGES
    for i, a in enumerate(sys.argv):
        if a == "--pages" and i + 1 < len(sys.argv):
            pages = int(sys.argv[i + 1])
            break

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    all_items = []

    print("广德市-环评审批: 爬取 %d 页" % pages)

    for idx in range(1, pages + 1):
        list_url = LIST_TPL % idx
        print("  [列表页 %d/%d] %s" % (idx, pages, list_url))
        html = fetch(list_url)
        if not html:
            continue

        items = extract_list_items(html)
        if not items:
            print("    -> 无数据，停止")
            break

        print("    -> 找到 %d 条" % len(items))

        for item_url, item_title, item_date in items:
            time.sleep(0.3)
            print("    [%d] %s..." % (len(all_items) + 1, item_title[:40]), end=" ", flush=True)
            detail_html = fetch(item_url)
            if not detail_html:
                print("fail")
                continue

            title, date, content, attachments = extract_detail(detail_html)
            if not title:
                title = item_title
            if not date:
                date = item_date

            all_items.append((item_url, title, date, content, attachments))
            status = "ok" if content else "empty"
            attach_msg = ""
            if attachments:
                attach_msg = " +%d附" % len(attachments)
            print("(%s, %d字%s)" % (status, len(content), attach_msg))

    if all_items:
        saved, skipped = push_to_db(conn, all_items)
        conn.commit()
        print("\n完成! 新增: %d, 跳过: %d, 总共: %d" % (saved, skipped, len(all_items)))

    conn.close()
    return 0


if __name__ == "__main__":
    sys.exit(main())
