#!/usr/bin/env python3
"""
南丰环境保护爬虫 (v2) - 使用 xxgk 信息公开搜索接口
栏目: col27107, infotypeId=C60000C60004C60001
"""
import sys, os, re, time, json, sqlite3, urllib.request, urllib.parse
from datetime import datetime, timedelta
import os

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.jxnf.gov.cn/module/xxgk/search.jsp"
SITE_NAME = "南丰环境保护"
COL_ID = "27107"
INFOTYPE_ID = "C60000C60004C60001"
PER_PAGE = 21  # 每页21条
THREE_YEARS_AGO = datetime.now() - timedelta(days=3*365)
THREE_YEAR_LIMIT = True  # 保留3年过滤

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Content-Type": "application/x-www-form-urlencoded"
}

def save_to_db(records):
    """写入 search.db 的 gov_raw 表"""
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        for r in records:
            db.execute("INSERT OR IGNORE INTO gov_raw(title, content, publish_date, source_url, page_url, site_name) VALUES (?,?,?,?,?,?)",
                       (r['title'], r['content'], r['publish_date'], r['url'], r['url'], SITE_NAME))
        db.commit()
    finally:
        db.close()

def clean_html(html_text):
    """清洗HTML转义"""
    text = html_text.replace('&amp;', '&').replace('&lt;', '<').replace('&gt;', '>').replace('&quot;', '"').replace('&#34;', '"').replace('&#39;', "'")
    text = re.sub(r'<br\s*/?>', '\n', text)
    text = re.sub(r'<p[^>]*>', '\n', text)
    text = re.sub(r'</p>', '', text)
    text = re.sub(r'<[^>]+>', '', text)
    return text.strip()

def fetch_page(page):
    """获取指定页码的列表"""
    post_data = {
        "infotypeId": INFOTYPE_ID,
        "jdid": "4",
        "area": "",
        "divid": "div7299",
        "vc_title": "",
        "vc_number": "",
        "currpage": str(page)
    }
    data = urllib.parse.urlencode(post_data).encode('utf-8')
    req = urllib.request.Request(BASE_URL, data=data, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        return html
    except Exception as e:
        print(f"  ⚠ 第{page}页请求失败: {e}")
        return ""

def parse_list(html, page):
    """解析列表页，提取文章信息"""
    items = []
    # 找所有列表项的链接
    # 格式: <li><a href='/art/...' title='xxx'>...</a><b>日期</b></li>
    pattern = re.compile(r"<li[^>]*>.*?<a[^>]*href=['\"]([^'\"]+)['\"][^>]*title=['\"]([^'\"]*)['\"][^>]*>.*?</a>\s*<b[^>]*>([^<]*)</b>", re.DOTALL)
    matches = pattern.findall(html)
    
    if not matches:
        # 另一种格式：没有<b>标签
        pattern2 = re.compile(r"<li[^>]*>.*?<a[^>]*href=['\"]([^'\"]+)['\"][^>]*title=['\"]([^'\"]*)['\"][^>]*>.*?</a>", re.DOTALL)
        matches2 = pattern2.findall(html)
        for url, title in matches2:
            # 尝试提取日期从URL或HTML
            date_match = re.search(r'(\d{4}/\d{1,2}/\d{1,2}|\d{4}-\d{1,2}-\d{1,2})', html)
            date_str = date_match.group(1) if date_match else ""
            items.append((url, title.strip(), date_str))
    else:
        items = [(url, title.strip(), date.strip()) for url, title, date in matches]
    
    return items

def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
def fetch_detail(url):
    """获取详情页正文 - 使用ZJEG_RSS.content标记"""
    full_url = url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        content = ""
        m = re.search(r'ZJEG_RSS\.content\.begin-->(.*?)<!--ZJEG_RSS\.content\.end', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<meta[^>]*>', '', content)
        else:
            m = re.search(r'文章正文开始\s*-->(.*?)<!--\s*bds button', html, re.DOTALL)
            if m:
                content = m.group(1).strip()
        return content.strip()
    except Exception as e:
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""
        print(f"  ⚠ 详情页失败: {url}: {e}")
        return ""

def main():
    import argparse as _AP
    _PARSER = _AP.ArgumentParser()
    _PARSER.add_argument("--pages", type=int, default=0, help="限制页数")
    _APG = _PARSER.parse_args()

    print(f"{'='*50}")
    print(f"{SITE_NAME} (col{COL_ID})")
    print(f"infotypeId={INFOTYPE_ID}")
    print(f"{'='*50}")
    
    # 先拿第一页获取总数和总页数
    html = fetch_page(1)
    if not html:
        print("❌ 无法获取第1页")
        return
    
    total_match = re.search(r'共(\d+)条', html)
    total_records = int(total_match.group(1)) if total_match else 0
    print(f"📊 总条数: {total_records}")
    
    total_pages = (total_records + PER_PAGE - 1) // PER_PAGE
    print(f"📄 总页数: {total_pages} (每页{PER_PAGE}条)")
    
    all_articles = []
    all_links = set()
    
    if _APG.pages and _APG.pages < total_pages:
        total_pages = _APG.pages
        print(f"🔧 限制页数: {total_pages}")

    for page in range(1, total_pages + 1):
        if page > 1:
            html = fetch_page(page)
            if not html:
                continue
        
        items = parse_list(html, page)
        print(f"  📄 第{page}页: {len(items)}条", end="")
        
        for url, title, date_str in items:
            # 去重
            if url in all_links:
                continue
            all_links.add(url)
            
            # 日期过滤（3年）
            pub_date = None
            if date_str:
                try:
                    pub_date = datetime.strptime(date_str.strip(), '%Y-%m-%d')
                except:
                    try:
                        pub_date = datetime.strptime(date_str.strip(), '%Y/%m/%d')
                    except:
                        pass
            
            if THREE_YEAR_LIMIT and pub_date and pub_date < THREE_YEARS_AGO:
                continue
            
            all_articles.append({
                'url': url if url.startswith('http') else f"http://www.jxnf.gov.cn{url}",
                'title': title,
                'publish_date': date_str,
                'pub_date_obj': pub_date
            })
        
        print(f" → 累计{len(all_articles)}条(去重)")
    
    print(f"\n📊 去重+3年过滤后: {len(all_articles)}条")
    print(f"\n{'='*50}")
    print("开始抓取详情页...")
    print(f"{'='*50}")
    
    batch = []
    total_saved = 0
    total_skipped = 0
    
    for i, article in enumerate(all_articles):
        title = article['title']
        url = article['url']
        date_str = article['publish_date']
        
        print(f"  [{i+1}/{len(all_articles)}] {title[:50]}...", end=" ")
        
        content = fetch_detail(url)
        if content:
            batch.append({
                'title': title,
                'content': content,
                'publish_date': date_str,
                'url': url
            })
            print(f"✅ ({len(content)}字)", end="")
            
            # 每10条入库一次
            if len(batch) >= 10:
                save_to_db(batch)
                total_saved += len(batch)
                batch = []
        else:
            print(f"⚠ 无正文", end="")
        
        print()
    
    # 入库剩余
    if batch:
        save_to_db(batch)
        total_saved += len(batch)
    
    print(f"\n{'='*50}")
    print(f"✅ 完成! 新增入库: {total_saved}条")
    
    # 更新FTS索引
    if total_saved > 0:
        db = sqlite3.connect(SEARCH_DB, timeout=60)
        try:
            db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) SELECT rowid, title, site_name, substr(content,1,200) FROM gov_raw WHERE site_name=? AND rowid NOT IN (SELECT rowid FROM gov_search WHERE rowid IN (SELECT rowid FROM gov_raw WHERE site_name=?))", (SITE_NAME, SITE_NAME))
            db.commit()
            print("   FTS索引已更新")
        except Exception as e:
            print(f"   FTS更新: {e}")
        finally:
            db.close()

if __name__ == '__main__':
    main()
