#!/usr/bin/env python3
"""宜阳县城市管理局-通知公告爬虫
只有第1页是静态HTML渲染，后续页paging API不可用
"""
import re, time, os, sqlite3
from datetime import datetime, date
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests

BASE = "https://www.yyzfw.gov.cn"
LIST_URL = "https://www.yyzfw.gov.cn/zwgk/zfjg/csglj/tzgg/"
CUTOFF_DATE = date(2023, 6, 17)
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "宜阳县-城市管理局-通知公告"
MAX_WORKERS = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}


def fetch(url):
    for i in range(3):
        try:
            r = requests.get(url, timeout=30, headers=HEADERS)
            r.encoding = "utf-8"
            if r.status_code == 200:
                return r.text
        except:
            if i < 2:
                time.sleep(2)
    return None


def extract_list_items(html):
    """从页面提取文章链接和日期"""
    items = []
    pattern = r'<a href="(https?://www.yyzfw.gov.cn/\d{4}/\d{2}-\d{2}/\d+\.html)"[^>]*>([^<]+)</a>'
    for m in re.finditer(pattern, html):
        url = m.group(1)
        title = m.group(2).strip()
        # 从URL提取日期
        dm = re.search(r'/(\d{4})/(\d{2})-(\d{2})/', url)
        pub_date = f"{dm.group(1)}-{dm.group(2)}-{dm.group(3)}" if dm else ""
        # 找链接后面的日期文本
        after = html[m.end():m.end()+50]
        dm2 = re.search(r'(\d{4}-\d{2}-\d{2})', after)
        if dm2:
            pub_date = dm2.group(1)
        items.append({"title": title, "url": url, "pub_date": pub_date})
    return items


def fetch_detail(detail_url):
    html = fetch(detail_url)
    if not html:
        return (detail_url, None, None, None)

    title = ""
    content = ""
    pub_date = ""

    m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    if m:
        title = m.group(1).strip()
    m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
    if m:
        pub_date = m.group(1).strip()[:10]
    m = re.search(r'<div class="mailbox_content_wznrs"[^>]*>(.*?)</div>\s*</div>\s*</div>', html, re.DOTALL)
    if not m:
        m = re.search(r'<div class="mailbox_content_wznrs"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()

    return (detail_url, title, content, pub_date)


def main():
    print("=== 宜阳县-城市管理局-通知公告 ===", flush=True)
    print("获取列表...", flush=True)

    html = fetch(LIST_URL)
    if not html:
        print("  无法获取列表页", flush=True)
        return

    items = extract_list_items(html)

    # 去重
    seen = set()
    unique_items = []
    for it in items:
        if it["url"] not in seen:
            seen.add(it["url"])
            unique_items.append(it)
    items = unique_items

    # 过滤近3年
    cutoff_items = []
    for it in items:
        try:
            d = date.fromisoformat(it["pub_date"])
            if d >= CUTOFF_DATE:
                cutoff_items.append(it)
        except:
            cutoff_items.append(it)

    print(f"  总计: {len(items)}条, 近3年: {len(cutoff_items)}条", flush=True)
    for it in cutoff_items[:5]:
        print(f"    {it['pub_date']} | {it['title'][:40]}", flush=True)

    if not cutoff_items:
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    cursor = conn.cursor()
    total_new = 0
    done = 0

    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        fut_map = {executor.submit(fetch_detail, it["url"]): it for it in cutoff_items}
        for fut in as_completed(fut_map):
            item = fut_map[fut]
            url, title, content, pub_date = fut.result()
            done += 1
            if done % 10 == 0:
                print(f"  详情 {done}/{len(cutoff_items)}...", flush=True)
            if title is None:
                continue
            if not title:
                title = item["title"]
            if not pub_date:
                pub_date = item["pub_date"]
            if not content:
                content = title
            try:
                cursor.execute("""
                    INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (SITE_NAME, title, url, content, pub_date, content,
                      int(datetime.strptime(pub_date, "%Y-%m-%d").timestamp())))
                if cursor.rowcount > 0:
                    total_new += 1
            except:
                pass
            conn.commit()

    cursor.execute("SELECT COUNT(*), MIN(publish_date), MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    cnt, min_d, max_d = cursor.fetchone()
    conn.close()

    print(f"\n=== 完成 ===", flush=True)
    print(f"  新增: {total_new}条", flush=True)
    print(f"  累计: {cnt}条 ({min_d} ~ {max_d})", flush=True)


if __name__ == "__main__":
    main()
