#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫：贵阳市生态环境局 - 项目环评审批公示
站点：https://sthjj.gz.gov.cn/hjgl/jsxm/hpspgg/
CMS：TRS CMS 标准分页模式
"""

import requests
import re
import sqlite3
import os
import sys
import time
from bs4 import BeautifulSoup
from urllib.parse import urljoin, urlparse
import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES

# ── 配置 ──
BASE_URL = "https://sthjj.gz.gov.cn/hjgl/jsxm/hpspgg/"
SITE_NAME = "贵阳市-项目环评审批"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Referer": "https://sthjj.gz.gov.cn/",
}
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
PAGE_SIZE = 12  # TRS 标准每页12条

# ── 工具函数 ──
def _get_db():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA synchronous=NORMAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn

def ensure_table(conn):
    conn.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY AUTOINCREMENT,
            title TEXT NOT NULL,
            content TEXT,
            source_url TEXT NOT NULL UNIQUE,
            publish_date TEXT,
            site_name TEXT,
            created_at TEXT DEFAULT (datetime('now','localtime'))
        )
    """)
    conn.execute("CREATE INDEX IF NOT EXISTS idx_gov_raw_site ON gov_raw(site_name)")
    conn.execute("CREATE INDEX IF NOT EXISTS idx_gov_raw_date ON gov_raw(publish_date)")
    # FTS 由 search.db 触发器 trg_gov_raw_fts_* 统一维护, 脚本不再自建 gov_fts 表
    conn.commit()

def row_exists(conn, source_url):
    cur = conn.execute("SELECT 1 FROM gov_raw WHERE source_url=?", (source_url,))
    return cur.fetchone() is not None

def insert_row(conn, title, content, source_url, publish_date):
    conn.execute(
        "INSERT OR IGNORE INTO gov_raw(title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
        (title, content, source_url, publish_date, SITE_NAME)
    )
    if conn.total_changes > 0:
        return True
    return False

def extract_content(soup):
    """从详情页提取正文"""
    # 尝试常见 TRS content 容器
    content_div = soup.select_one("#content, .content, .article-content, .info-content, .TRS_Editor")
    if content_div:
        # 处理图片
        for img in content_div.find_all("img"):
            src = img.get("src", "")
            if src and not src.startswith("data:"):
                img_url = urljoin(BASE_URL, src) if not src.startswith("http") else src
                img.replace_with(f'<img src="{img_url}">')
            else:
                img.decompose()
        return str(content_div)
    
    # 备选：取 body 主要部分
    body = soup.find("body")
    if body:
        return str(body)
    return ""

def get_detail(url, session, retries=3):
    """获取详情页内容"""
    for attempt in range(retries):
        try:
            r = session.get(url, headers=HEADERS, timeout=60)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                soup = BeautifulSoup(r.text, 'html.parser')
                content = extract_content(soup)
                
                # 尝试提取日期
                date_str = ""
                text = r.text
                date_patterns = [
                    r'(\d{4}[-/\.年]\d{1,2}[-/\.月]\d{1,2}日?)',
                    r'发布时间[：:]\s*(\d{4}[-/\.]\d{1,2}[-/\.]\d{1,2})',
                    r'发布日期[：:]\s*(\d{4}[-/\.]\d{1,2}[-/\.]\d{1,2})',
                    r'(\d{4}-\d{2}-\d{2})',
                ]
                for p in date_patterns:
                    m = re.search(p, text)
                    if m:
                        date_str = m.group(1).replace('年','-').replace('月','-').replace('日','').replace('/','-').replace('.','-')
                        break
                
                return content, date_str
        except requests.RequestException as e:
            if attempt < retries - 1:
                time.sleep(2 ** attempt)
    return "", ""

def get_total_pages():
    """获取总页数"""
    try:
        r = requests.get(BASE_URL, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            print(f"无法访问首页: HTTP {r.status_code}")
            return 0
        # 找总记录数
        m = re.search(r'记录总数[：:]\s*(\d+)', r.text)
        if m:
            total = int(m.group(1))
            pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
            print(f"首页检测到总记录: {total}条, 共{pages}页")
            return pages
        # 找最后一页链接: index_N.htm
        soup = BeautifulSoup(r.text, 'html.parser')
        page_links = soup.select('a[href*="index_"]')
        max_page = 1
        for a in page_links:
            href = a.get('href', '')
            m2 = re.search(r'index_(\d+)\.htm', href)
            if m2:
                p = int(m2.group(1))
                if p > max_page:
                    max_page = p
        print(f"首页检测到最大页: {max_page}")
        return max_page
    except Exception as e:
        print(f"获取总页数失败: {e}")
        return 0

def crawl():
    """全量采集"""
    conn = _get_db()
    ensure_table(conn)
    
    total_pages = get_total_pages()
    if total_pages == 0:
        print("无法确定总页数，退出")
        conn.close()
        return
    
    print(f"\n共 {total_pages} 页，开始采集...")
    
    session = requests.Session()
    session.headers.update(HEADERS)
    
    total_new = 0
    total_skip = 0
    
    for page in range(1, min(total_pages, _MAX_PG or total_pages) + 1):
        if page == 1:
            url = BASE_URL
        else:
            url = f"https://sthjj.gz.gov.cn/hjgl/jsxm/hpspgg/index_{page-1}.htm"
        
        try:
            r = session.get(url, timeout=60)
            r.encoding = 'utf-8'
            if r.status_code != 200:
                print(f"  第{page}页 HTTP {r.status_code}，跳过")
                continue
        except Exception as e:
            print(f"  第{page}页请求失败: {e}")
            continue
        
        soup = BeautifulSoup(r.text, 'html.parser')
        
        # 查找列表链接 - TRS 标准模式
        links_found = 0
        for a in soup.find_all('a', href=True):
            href = a['href']
            title = a.get('title', '') or a.get_text(strip=True)
            
            # 过滤：必须是详情页链接（含 .htm 或 .html，非分页链接）
            if not (href.endswith('.htm') or href.endswith('.html') or '/content' in href):
                continue
            if 'index_' in href or href.endswith('index.htm') or href.endswith('index.html'):
                continue
            if not title or len(title) < 5:
                continue
            
            # 跳过纯数字/日期类标题
            if re.match(r'^\d[\d\.\s\-]*$', title):
                continue
            
            detail_url = urljoin(url, href)
            
            links_found += 1
            if row_exists(conn, detail_url):
                total_skip += 1
                continue
            
            # 获取详情
            content, date_str = get_detail(detail_url, session)
            
            # 验证内容
            content_text = BeautifulSoup(content, 'html.parser').get_text(strip=True) if content else ""
            if len(content_text) < 20:
                print(f"  ⚠ 内容过短: {title[:30]}... 跳过")
                total_skip += 1
                continue
            
            if insert_row(conn, title, content, detail_url, date_str):
                total_new += 1
                conn.commit()
                print(f"  ✓ [{total_new}] {title[:40]}... ({date_str or '无日期'})")
            else:
                total_skip += 1
        
        page_total = total_new + total_skip
        print(f"  第{page}/{total_pages}页: 本页{links_found}条, 累计新{total_new}条/跳过{total_skip}条")
        
        # 每页间隔
        time.sleep(1)
    
    conn.close()
    print(f"\n✅ 采集完成！新增{total_new}条，跳过{total_skip}条，总处理{total_new+total_skip}条")


if __name__ == '__main__':
    crawl()
