#!/usr/bin/env python3
"""crawl_ganyu_haitou.py — 赣榆区海头镇人民政府-政府文件"""

import os, re, sqlite3, time, random, urllib.parse
from datetime import datetime, timedelta

DB_PATH = os.environ.get('SEARCH_DB', '/root/search.db')
SITE_NAME = '赣榆区海头镇人民政府-政府文件'
LIST_URL = 'https://www.ganyu.gov.cn/gyqhtzrmzf/zfwj/zfwj.html'
BASE = 'https://www.ganyu.gov.cn'
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36',
    'Accept-Language': 'zh-CN,zh;q=0.9',
}

import requests
sess = requests.Session()
sess.headers.update(HEADERS)


def fetch(url, timeout=15):
    for _ in range(3):
        try:
            r = sess.get(url, timeout=timeout)
            r.encoding = 'utf-8'
            return r.text
        except Exception as e:
            time.sleep(1)
    return None


def parse_list(html):
    """解析 #initData 中全部 18 条记录"""
    m = re.search(r'<div\s+id="initData"[^>]*>(.*?)</div>\s*<!--列表内容结束-->', html, re.DOTALL)
    if not m:
        return []
    records = []
    for ul in re.finditer(r'<ul\s+class="xxgk-list-con"[^>]*>(.*?)</ul>', m.group(1), re.DOTALL):
        li = re.search(r'<li>(.*?)</li>', ul.group(1), re.DOTALL)
        if not li:
            continue
        a = re.search(r'<a\s+href="([^"]*)"', li.group(1))
        span = re.search(r'<span>([^<]+)</span>', li.group(1))
        title_a = re.search(r'title="([^"]*)"', li.group(1))
        if not a or not span:
            continue
        records.append((
            title_a.group(1).strip() if title_a else '',
            urllib.parse.urljoin(BASE, a.group(1)),
            span.group(1).strip()
        ))
    return records


def parse_detail(html):
    """提取标题 + 正文"""
    t = re.search(r'<div\s+class="art-title">\s*(.*?)\s*</div>', html, re.DOTALL)
    title = t.group(1).strip() if t else ''
    m = re.search(r'<div\s+class="art-main"\s+id="zoom"[^>]*>(.*?)</div>\s*<div\s+class="art-print"', html, re.DOTALL)
    if not m:
        m = re.search(r'<div\s+class="art-main"\s+id="zoom"[^>]*>(.*?)</div>', html, re.DOTALL)
    content = m.group(1).strip() if m else ''
    # 清理
    content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
    content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
    content = re.sub(r'<p>\s*(?:<br\s*/?>\s*)*</p>', '', content)
    return title, content.strip()


def main():
    print(f'[{datetime.now().strftime("%H:%M:%S")}] {SITE_NAME}')

    html = fetch(LIST_URL)
    if not html:
        print('[ERROR] 列表页获取失败')
        return

    records = parse_list(html)
    print(f'共 {len(records)} 条记录')

    recent = [(t, u, d) for t, u, d in records if d >= CUTOFF]
    print(f'近3年: {len(recent)} 条 (过滤 {len(records) - len(recent)} 条)')

    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    new = skip = err = 0

    for i, (title, url, pub_date) in enumerate(recent, 1):
        print(f'  [{i}/{len(recent)}] {title[:30]}...', end=' ')

        dup = c.execute(
            'SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?',
            (url, SITE_NAME)
        ).fetchone()
        if dup:
            print('DUPLICATE')
            skip += 1
            continue

        dh = fetch(url)
        if not dh:
            print('FETCH ERR')
            err += 1
            continue

        dt, content = parse_detail(dh)
        final_title = dt or title

        try:
            c.execute('''INSERT OR IGNORE INTO gov_raw
                (site_name, title, page_url, publish_date, source_url, content)
                VALUES (?,?,?,?,?,?)''',
                (SITE_NAME, final_title, url, pub_date, url, content))
            if c.rowcount:
                new += 1
                print(f'OK ({len(content)}B)')
            else:
                print('DUP(insert)')
                skip += 1
        except Exception as e:
            print(f'ERR: {e}')
            err += 1

        conn.commit()
        time.sleep(random.uniform(0.3, 0.5))

    conn.close()
    print(f'\n✅ 新入库: {new}  跳过: {skip}  错误: {err}')


if __name__ == '__main__':
    main()
