#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫：广州市生态环境局 - 环评审批前公示
站点：https://sthjj.gz.gov.cn/hjgl/jsxm/hpspqgs/
"""

import requests
import re
import sqlite3
import os
import time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://sthjj.gz.gov.cn/hjgl/jsxm/hpspqgs/"
SITE_NAME = "广州市-环评审批前公示"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36"}
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 100

def _get_db():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA synchronous=NORMAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn

def ensure_table(conn):
    conn.execute("CREATE TABLE IF NOT EXISTS gov_raw (id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT NOT NULL, content TEXT, source_url TEXT NOT NULL UNIQUE, publish_date TEXT, site_name TEXT, created_at TEXT DEFAULT (datetime('now','localtime')))")
    conn.execute("CREATE INDEX IF NOT EXISTS idx_gov_raw_site ON gov_raw(site_name)")
    conn.execute("CREATE INDEX IF NOT EXISTS idx_gov_raw_date ON gov_raw(publish_date)")
    conn.execute("CREATE VIRTUAL TABLE IF NOT EXISTS gov_search_v3 USING fts5(title, content, source_url, publish_date, site_name, tokenize='trigram')")
    conn.commit()

def row_exists(conn, source_url):
    cur = conn.execute("SELECT 1 FROM gov_raw WHERE source_url=?", (source_url,))
    return cur.fetchone() is not None

def insert_row(conn, title, content, source_url, publish_date):
    conn.execute("INSERT OR IGNORE INTO gov_raw(title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                 (title, content, source_url, publish_date, SITE_NAME))
    if conn.total_changes > 0:
        try:
            conn.execute("INSERT INTO gov_search_v3(title, content, source_url, publish_date, site_name) VALUES (?,?,?,?,?)",
                         (title, content, source_url, publish_date, SITE_NAME))
        except Exception:
            pass
        return True
    return False

def extract_content(soup):
    div = soup.select_one("#logPanel")
    if div:
        for img in div.find_all("img"):
            src = img.get("src", "")
            if src and not src.startswith("data:"):
                img.replace_with(f'<img src="{"https://sthjj.gz.gov.cn" + src if not src.startswith("http") else src}">')
            else:
                img.decompose()
        return str(div)
    return ""

def extract_title(soup):
    for sel in ['.h1', 'h1', '.title']:
        el = soup.select_one(sel)
        if el and 5 <= len(el.get_text(strip=True)) <= 200:
            return el.get_text(strip=True)
    return ""

def get_detail(url, session, retries=3):
    for attempt in range(retries):
        try:
            r = session.get(url, headers=HEADERS, timeout=60)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                soup = BeautifulSoup(r.text, 'html.parser')
                return extract_content(soup), extract_title(soup)
        except requests.RequestException as e:
            if attempt < retries - 1:
                time.sleep(2 ** attempt)
    return "", ""

def parse_list_page(url, session):
    results = []
    try:
        r = session.get(url, headers=HEADERS, timeout=60)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return results
    except Exception:
        return results
    soup = BeautifulSoup(r.text, 'html.parser')
    for a in soup.find_all('a', href=True):
        h = a['href']
        if 'content/post_' not in h:
            continue
        date_span = a.find_parent('div', class_='conts-list')
        date_str = ""
        if date_span:
            spans = date_span.find_all('span')
            if len(spans) >= 2:
                ds = spans[-1].get_text(strip=True)
                if re.match(r'\d{4}-\d{2}-\d{2}', ds):
                    date_str = ds
        title = a.get_text(strip=True)
        if not title or len(title) < 5 or '穗好办' in title:
            continue
        detail_url = h if h.startswith('http') else 'https://sthjj.gz.gov.cn' + h if h.startswith('/') else urljoin(url, h)
        results.append((title, detail_url, date_str))
    return results

def crawl():
    conn = _get_db()
    ensure_table(conn)
    session = requests.Session()
    session.headers.update(HEADERS)
    total_new = 0
    total_skip = 0
    for page in range(1, MAX_PAGES + 1):
        url = BASE_URL if page == 1 else f"{BASE_URL}index_{page}.html"
        articles = parse_list_page(url, session)
        if not articles:
            print(f"  第{page}页无文章，结束")
            break
        for list_title, detail_url, date_str in articles:
            if row_exists(conn, detail_url):
                total_skip += 1
                continue
            content, proper_title = get_detail(detail_url, session)
            final_title = proper_title if proper_title else list_title
            ct = BeautifulSoup(content or '', 'html.parser').get_text(strip=True)
            if len(ct) < 20:
                total_skip += 1
                continue
            if insert_row(conn, final_title, content, detail_url, date_str):
                total_new += 1
                conn.commit()
                print(f"  ✓ [{total_new}] {final_title[:40]} ({date_str})")
            else:
                total_skip += 1
        print(f"  第{page}/{MAX_PAGES}页: {len(articles)}条, 累计新{total_new}条/跳过{total_skip}条")
        time.sleep(1)
    conn.close()
    print(f"\n✅ 完成！新增{total_new}条，跳过{total_skip}条")

if __name__ == '__main__':
    crawl()
