#!/usr/bin/env python3
"""岳麓高新区-行政许可公示 爬虫"""
import requests
import re
import os
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "岳麓高新区-行政许可公示"
BASE_URL = "https://ylgxq.changsha.gov.cn"
LIST_PATH = "/xxgk/fdxxgknr/xzxk_138208/xzxkgs"
MAX_PAGES = 30

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
}


def fetch(url):
    """带SSL跳过和浏览器头获取页面"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  ❌ 请求失败: {e}")
        return None


def parse_list_page(html):
    """解析列表页，返回 [(title, url, date_str), ...]"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    # 列表在 div.xxgk-list > ul > li
    lst_div = soup.find("div", class_="xxgk-list")
    if not lst_div:
        return items
    ul = lst_div.find("ul")
    if not ul:
        return items
    for li in ul.find_all("li", recursive=False):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        title = a.get_text(strip=True)
        if not href or not title:
            continue
        # Relative URL like ./202606/t20260616_12408189.html
        if href.startswith("./"):
            detail_url = BASE_URL + LIST_PATH + "/" + href[2:]
        elif href.startswith("http"):
            detail_url = href
        elif href.startswith("/"):
            detail_url = BASE_URL + href
        else:
            detail_url = BASE_URL + LIST_PATH + "/" + href

        # Date from <span>
        date_span = li.find("span")
        date_str = date_span.get_text(strip=True) if date_span else ""

        items.append((title, detail_url, date_str))
    return items


def parse_detail(html):
    """解析详情页，返回 (title, content, publish_date)"""
    soup = BeautifulSoup(html, "html.parser")

    # Title from <h2 class="art-tit">
    title = ""
    h2 = soup.find("h2", class_="art-tit")
    if h2:
        title = h2.get_text(strip=True)

    # Date
    publish_date = ""
    # Try <p class="fl">发布时间 : 2026-06-16</p>
    info_div = soup.find("div", class_="info")
    if info_div:
        fl_p = info_div.find("p", class_="fl")
        if fl_p:
            text = fl_p.get_text()
            m = re.search(r'(\d{4}-\d{2}-\d{2})', text)
            if m:
                publish_date = m.group(1)

    if not publish_date:
        # Try meta tag with spaces in name
        meta = soup.find("meta", attrs={"name": " Pubdate "})
        if meta and meta.get("content"):
            m = re.search(r'(\d{4}-\d{2}-\d{2})', meta["content"])
            if m:
                publish_date = m.group(1)

    # Content from <div class="content-main">
    content = ""
    content_div = soup.find("div", class_="content-main")
    if content_div:
        content = str(content_div)

    return title, content, publish_date


def crawl():
    conn = None
    try:
        import sqlite3
        conn = sqlite3.connect(SEARCH_DB, timeout=60)
        c = conn.cursor()
        c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            title TEXT,
            content TEXT,
            source_url TEXT UNIQUE,
            site_name TEXT,
            publish_date TEXT
        )''')

        total_new = 0
        total_skipped = 0

        for page in range(1, MAX_PAGES + 1):
            if page == 1:
                page_url = BASE_URL + LIST_PATH + "/index.html"
            else:
                page_url = BASE_URL + LIST_PATH + f"/index_{page}.html"

            html = fetch(page_url)
            if not html:
                continue

            items = parse_list_page(html)
            if not items:
                print(f"第{page}/{MAX_PAGES}页: 无数据，结束分页")
                break

            print(f"第{page}页: {len(items)}条")

            for i, (title, detail_url, date_from_list) in enumerate(items, 1):
                # Check if exists
                c.execute("SELECT id FROM gov_raw WHERE source_url = ?", (detail_url,))
                if c.fetchone():
                    total_skipped += 1
                    continue

                # Fetch detail
                detail_html = fetch(detail_url)
                if not detail_html:
                    print(f"  [{i}/{len(items)}] ❌ 详情失败: {title[:30]}...")
                    continue

                _, content, publish_date = parse_detail(detail_html)
                if not publish_date:
                    publish_date = date_from_list

                try:
                    c.execute(
                        "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                        (title, content, detail_url, SITE_NAME, publish_date)
                    )
                    conn.commit()
                    total_new += 1
                    content_len = len(content) if content else 0
                    status = " ⚠️ 内容过短" if content_len < 200 else ""
                    print(f"  [{i}/{len(items)}] ✅ {title[:35]}... ({publish_date}){status}")
                except Exception as e:
                    print(f"  [{i}/{len(items)}] ❌ 入库失败: {e}")

        print(f"\n✅ 爬取完成！新增: {total_new}, 跳过(已存在): {total_skipped}")
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
        total = c.fetchone()[0]
        print(f"  共 {total} 条")

    finally:
        if conn:
            conn.close()


if __name__ == "__main__":
    crawl()
