#!/usr/bin/env python3
"""宜都市人民政府 - 公示公告 爬虫
List API: xxgkapi.yichang.gov.cn/show/lists (recommend=9, 36页, 713条)
Detail API: xxgkapi.yichang.gov.cn/show/detail
"""
import re, requests
from datetime import datetime
import sqlite3, os

API_BASE = "https://xxgkapi.yichang.gov.cn"
SITE_NAME = "yidu.gov.cn-公示公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CATEGORY = "eia"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
CUTOFF_DATE = "2023-06-16"


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.row_factory = sqlite3.Row
    return conn


def clean_text(text):
    if not text:
        return ""
    return re.sub(r'\s+', ' ', text.strip())


def extract_detail(session, article_id):
    """获取详情"""
    try:
        resp = session.get(f"{API_BASE}/show/detail", params={
            "areaid": 8, "id": article_id, "cache": "on"
        }, headers=HEADERS, timeout=30)
        data = resp.json()
        if isinstance(data, list) and len(data) > 0:
            item = data[0]
        else:
            return None, None, None, None
    except Exception as e:
        print(f"  [WARN] 获取详情失败 id={article_id}: {e}", flush=True)
        return None, None, None, None

    title = clean_text(item.get('title', ''))
    if not title:
        return None, None, None, None

    # 日期
    pub_date = ''
    vc_time = item.get('vc_inputtime', '')
    m = re.match(r'(\d{4}-\d{2}-\d{2})', vc_time)
    if m:
        pub_date = m.group(1)
    if not pub_date:
        return None, None, None, None

    # 正文 + 附件
    content = item.get('content', '') or ''
    fujian = item.get('fujian', '') or ''
    if fujian:
        content += f'\n<p><a href="{fujian}" target="_blank">附件下载</a></p>'

    # 摘要
    summary = ''
    if content:
        text_obj = clean_text(content[:500])
        summary = text_obj[:200]

    return title, pub_date, content, summary


def main():
    print(f"=== {SITE_NAME} 爬虫 ===", flush=True)
    print(f"数据库: {DB_PATH}", flush=True)

    conn = get_conn()
    cur = conn.cursor()
    cutoff_dt = datetime.strptime(CUTOFF_DATE, "%Y-%m-%d").date()

    session = requests.Session()

    # 列表API - 获取所有文章
    all_articles = []
    base_params = {
        "areaid": 8, "webid": 998, "recommend": 9,
        "page": 1, "pagenums": 20, "orderby": 0, "vc_title": ""
    }

    for page in range(1, 40):
        params = {**base_params, "page": page}
        try:
            resp = session.get(f"{API_BASE}/show/lists", params=params, headers=HEADERS, timeout=30)
            data = resp.json()
            items = data.get('lists', [])
            if not items:
                break
            for item in items:
                pub_date = ''
                m = re.match(r'(\d{4}-\d{2}-\d{2})', item.get('vc_inputtime', ''))
                if m:
                    pub_date = m.group(1)
                all_articles.append({
                    'id': item['n_id'],
                    'title': clean_text(item.get('title', '')),
                    'pub_date': pub_date,
                })
            print(f"  第{page}页: {len(items)}条", flush=True)
        except Exception as e:
            print(f"  [ERROR] 第{page}页API失败: {e}", flush=True)
            break

    print(f"\n共获取 {len(all_articles)} 条列表条目", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for art in all_articles:
        aid = art['id']
        detail_url = f"{API_BASE}/show/detail?id={aid}"

        # 查重
        cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (detail_url,))
        if cur.fetchone():
            total_skip += 1
            continue

        # 日期过滤
        pub_date = art['pub_date']
        if pub_date:
            pub_dt = datetime.strptime(pub_date, "%Y-%m-%d").date()
            if pub_dt < cutoff_dt:
                total_skip += 1
                continue

        # 获取详情
        print(f"  [{total_new+total_skip+total_error+1}/{len(all_articles)}] id={aid}", flush=True)
        title, detail_date, content, summary = extract_detail(session, aid)

        if not title:
            print(f"  [SKIP] 无详情: {art['title'][:40]}", flush=True)
            total_error += 1
            continue

        date_rank = int(pub_date.replace('-', ''))
        source_url = "https://www.yidu.gov.cn/list-41-1.html"

        try:
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (page_url, title, content, site_name, publish_date, summary, category, date_rank, source_url) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
                (detail_url, title, content, SITE_NAME, pub_date, summary, CATEGORY, date_rank, source_url)
            )
            conn.commit()
            total_new += 1
            print(f"  [OK] {title[:40]}... | {pub_date}", flush=True)
        except Exception as e:
            print(f"  [ERROR] 入库失败: {title[:30]}... - {e}", flush=True)
            conn.rollback()
            total_error += 1

    conn.close()
    print(f"\n=== 完成: 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error} ===", flush=True)


if __name__ == '__main__':
    main()
