#!/usr/bin/env python3
"""嘉鱼县人民政府 - 政府信息公开年度报告 爬虫"""
import re, urllib.request, sys
from datetime import datetime

DB_PATH = "/root/search.db"
BASE_URL = "http://www.jiayu.gov.cn/xxgk/xxgknb"
SITE_NAME = "jiayu_xxgknb"
MAX_PAGES = 5  # 首次最多5页，日跑仅1页

def fetch(url, encoding='utf-8'):
    req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'})
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        return resp.read().decode(encoding, errors='replace')
    except Exception as e:
        print(f"请求失败 {url}: {e}", flush=True)
        return None

def extract_content(html):
    """提取 article-box#fontzoom > .TRS_UEDITOR 内的正文"""
    m = re.search(r'<div class="article-box"[^>]* id="fontzoom"[^>]*>(.*?)</div>\s*(?:<div class="article-enclosure"|<div class="clear)', html, re.S)
    if m:
        content = m.group(1)
        return content.strip()
    m2 = re.search(r'<div class="view[^>]*>(.*?)</div>\s*', html, re.S)
    if m2:
        return m2.group(1).strip()
    return ""

def extract_text(html_content):
    if not html_content:
        return ""
    text = re.sub(r'<[^>]+>', ' ', html_content)
    text = re.sub(r'\s+', ' ', text).strip()
    return text[:200]

def crawl():
    now = datetime.now()
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = c.fetchone()[0]
    incremental = existing > 0
    max_pages = 1 if incremental else MAX_PAGES
    total_new = 0
    total_skip = 0

    for page in range(max_pages):
        if page == 0:
            list_url = f"{BASE_URL}/index.shtml"
        else:
            list_url = f"{BASE_URL}/index_{page}.shtml"

        print(f"正在爬取第{page+1}页: {list_url}", flush=True)
        html = fetch(list_url)
        if not html:
            continue

        # 解析列表: <li><a href="..." title="标题">标题</a><span>日期</span></li>
        items = re.findall(
            r"<li>.*?<a href='([^']+)'[^>]*title='([^']*)'[^>]*>(.*?)</a>\s*<span>([^<]+)</span>",
            html, re.DOTALL
        )
        # 也试试双引号
        if not items:
            items = re.findall(
                r'<li>.*?<a href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span>([^<]+)</span>',
                html, re.DOTALL
            )

        if not items:
            print(f"第{page+1}页无数据", flush=True)
            continue

        print(f"本页{len(items)}条", flush=True)

        for href, title_attr, link_text, pub_date in items:
            title = title_attr.strip() or link_text.strip()
            if not title:
                continue
            pub_date = pub_date.strip()

            # 处理相对URL
            if href.startswith('http'):
                detail_url = href
            elif href.startswith('/'):
                detail_url = f"http://www.jiayu.gov.cn{href}"
            else:
                detail_url = f"{BASE_URL}/{href}"

            # 检查重复
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
            if c.fetchone():
                total_skip += 1
                continue

            # 获取详情
            detail_html = fetch(detail_url)
            if not detail_html:
                content = ""
                summary = title[:200]
            else:
                content = extract_content(detail_html)
                summary = extract_text(content)

            try:
                c.execute(
                    "INSERT INTO gov_raw (site_name, page_url, title, publish_date, summary, content, category, source_url) VALUES (?,?,?,?,?,?,?,?)",
                    (SITE_NAME, detail_url, title, pub_date, summary or title[:200], content, '政府信息公开年度报告',
                     "http://www.jiayu.gov.cn/xxgk/xxgknb/index.shtml")
                )
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_skip += 1
            except Exception as e:
                print(f"入库失败 [{title[:30]}]: {e}", flush=True)
                conn.rollback()

        print(f"第{page+1}页完成，新增累计{total_new}条，跳过{total_skip}条", flush=True)

    conn.close()
    print(f"\n嘉鱼县信息公开年报爬取完成，共新增 {total_new} 条，跳过 {total_skip} 条", flush=True)

if __name__ == '__main__':
    crawl()
