#!/usr/bin/env python3
"""南城县人民政府 - 环评信息 爬虫
List: POST /module/xxgk/search.jsp (53 pages, 941 records)
Detail: /art/YYYY/M/D/art_27230_XXXXXX.html
"""
import re, requests
from bs4 import BeautifulSoup
from datetime import datetime
from urllib.parse import urljoin
import sqlite3, os

BASE_URL = "https://www.jxnc.gov.cn"
SITE_NAME = "jxnc.gov.cn-环评信息"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CATEGORY = "eia"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
CUTOFF_DATE = "2023-06-16"
PER_PAGE = 18


def get_conn():
    conn = sqlite3.connect(DB_PATH)
    conn.row_factory = sqlite3.Row
    return conn


def clean_text(text):
    if not text:
        return ""
    return re.sub(r'\s+', ' ', text.strip())


def get_meta(html, name):
    m = re.search(rf'<meta[^>]+name=["\']{name}["\'][^>]*content=["\']([^"\']+)["\']', html)
    return m.group(1) if m else ''


def extract_detail(url):
    """获取详情页 - 标题/日期/正文"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        resp.encoding = 'utf-8'
    except Exception as e:
        print(f"  [WARN] 获取详情页失败 {url}: {e}", flush=True)
        return None, None, None, None

    html = resp.text
    soup = BeautifulSoup(html, 'html.parser')

    # 标题
    title = get_meta(html, 'ArticleTitle')
    if not title:
        h2 = soup.select_one('h2, .bt-article-y h2, .bt-article-02 h2, .artile_zw h2')
        if h2:
            title = h2.get_text(strip=True)
    if not title:
        return None, None, None, None
    title = clean_text(title)

    # 日期
    pub_date = get_meta(html, 'PubDate')
    pub_date_m = re.match(r'(\d{4}-\d{2}-\d{2})', pub_date)
    pub_date = pub_date_m.group(1) if pub_date_m else None
    if not pub_date:
        pub_m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
        if pub_m:
            pub_date = pub_m.group(1)
    if not pub_date:
        return None, None, None, None

    # 来源
    source = get_meta(html, 'ContentSource')

    # 正文 - multiple possible containers
    content = ''
    for selector in ['#zoom', '#barrierfree_container', '.bt-article-y.bfr_article_content', '.artile_zw', '.bt-article-02', '.content-box', '.article-content', '.con_text', '.TRS_Editor', '.Custom_UnionStyle', '#content']:
        el = soup.select_one(selector)
        if el:
            txt = el.get_text(strip=True)
            if len(txt) > 50:
                # Keep attachments (PDF links etc)
                content = str(el)
                break

    summary = ''
    if content:
        text_obj = BeautifulSoup(content, 'html.parser')
        text = text_obj.get_text(separator=' ', strip=True)
        summary = clean_text(text[:200])

    return title, pub_date, content, summary


def fetch_page(session, page):
    """获取单页列表"""
    data = f"infotypeId=N00004N00004N00010N00003&jdid=53&area=&divid=div7299&vc_title=&vc_number=&currpage={page}&sortfield="
    try:
        resp = session.post(BASE_URL + "/module/xxgk/search.jsp", data=data, headers={
            **HEADERS,
            "Content-Type": "application/x-www-form-urlencoded",
            "X-Requested-With": "XMLHttpRequest",
            "Referer": f"{BASE_URL}/col/col27230/index.html?number=N00004N00004N00010N00003",
        }, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f"  [ERROR] API请求第{page}页失败: {e}", flush=True)
        return [], 0

    # 解析文章列表
    # <li><a href='http://...art_27230_XXXXXX.html' target='_blank' title="标题">标题</a><b>日期</b></li>
    articles = []
    # Match li items with article links
    for m in re.finditer(
        r"<li[^>]*>\s*<a[^>]*href=['\"]([^'\"]+)['\"][^>]*title=['\"]([^'\"]+)['\"][^>]*>(.*?)</a>\s*<b[^>]*>\s*([^<]+)\s*</b>\s*</li>",
        html, re.DOTALL
    ):
        href = m.group(1).strip()
        title = clean_text(m.group(2))
        date_str = clean_text(m.group(4))
        date_m = re.match(r'(\d{4}-\d{2}-\d{2})', date_str)
        pub_date = date_m.group(1) if date_m else None
        articles.append({
            'title': title,
            'url': href if href.startswith('http') else urljoin(BASE_URL, href),
            'pub_date': pub_date,
        })

    # 总记录数
    total_m = re.search(r'共(\d+)条记录', html)
    total_records = int(total_m.group(1)) if total_m else len(articles)

    return articles, total_records


def main():
    print(f"=== {SITE_NAME} 爬虫 ===", flush=True)
    print(f"数据库: {DB_PATH}", flush=True)

    conn = get_conn()
    cur = conn.cursor()
    cutoff_dt = datetime.strptime(CUTOFF_DATE, "%Y-%m-%d").date()

    session = requests.Session()
    # 初始化会话
    session.get(f"{BASE_URL}/col/col27230/index.html?number=N00004N00004N00010N00003", headers=HEADERS, timeout=15)

    # 获取第一页，获取总记录数
    first_articles, total_records = fetch_page(session, 1)
    total_pages = (total_records + PER_PAGE - 1) // PER_PAGE
    print(f"总记录: {total_records}, 每页{PER_PAGE}条, 共{total_pages}页", flush=True)

    all_articles = list(first_articles)

    for page in range(2, total_pages + 1):
        print(f"\n--- 第 {page}/{total_pages} 页 ---", flush=True)
        articles, _ = fetch_page(session, page)
        if not articles:
            print("  无内容，停止分页", flush=True)
            break
        all_articles.extend(articles)

    print(f"\n共获取 {len(all_articles)} 条列表条目", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for art in all_articles:
        url = art['url']

        # 查重
        cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
        if cur.fetchone():
            total_skip += 1
            continue

        # 日期过滤
        pub_date = art.get('pub_date')
        if pub_date:
            pub_dt = datetime.strptime(pub_date, "%Y-%m-%d").date()
            if pub_dt < cutoff_dt:
                total_skip += 1
                continue

        # 获取详情
        print(f"  [{total_new+total_skip+total_error+1}/{len(all_articles)}] {url.split('/')[-1]}", flush=True)
        title, detail_date, content, summary = extract_detail(url)

        if not pub_date and detail_date:
            pub_date = detail_date
        if not pub_date:
            print(f"  [SKIP] 无日期: {art['title'][:40]}...", flush=True)
            total_error += 1
            continue

        if title:
            pass  # use detail title
        else:
            title = art['title']

        date_rank = int(pub_date.replace('-', ''))
        source_url = f"{BASE_URL}/col/col27230/index.html?number=N00004N00004N00010N00003"

        try:
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category, date_rank, source_url) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
                (url, title, content, SITE_NAME, pub_date, summary, CATEGORY, date_rank, source_url)
            )
            conn.commit()
            total_new += 1
            print(f"  [OK] {title[:40]}... | {pub_date}", flush=True)
        except Exception as e:
            print(f"  [ERROR] 入库失败: {title[:30]}... - {e}", flush=True)
            conn.rollback()
            total_error += 1

    conn.close()
    print(f"\n=== 完成: 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error} ===", flush=True)


if __name__ == '__main__':
    main()
