#!/usr/bin/env python3
"""益阳市生态环境局 - 环保影响评价 爬虫 (CreatorCMS)"""
import re, sys, os, urllib.request, urllib.parse, urllib.error
from datetime import datetime

DB_PATH = "/root/search.db"
BASE_URL = "https://www.yiyang.gov.cn/yyshjbhj/3452/3467"
SITE_NAME = "yiyang_shj"
MAX_PAGES = 15

def fetch(url, encoding='gb2312'):
    req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'})
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read()
        return html.decode(encoding, errors='replace')
    except Exception as e:
        print(f"请求失败 {url}: {e}", flush=True)
        return None

def extract_content(html):
    """提取 detail-text#zoom 内的正文"""
    m = re.search(r'<div class="detail-text"[^>]* id="zoom"[^>]*>(.*?)</div>\s*(?:<!--.*?-->)?\s*(?:<div[^>]*id="div_div"|</wbr>|$)', html, re.S)
    if m:
        content = m.group(1)
        # 清理 <wbr> 标签
        content = content.replace('<wbr>', '').replace('</wbr>', '')
        return content.strip()
    # 备用: 取大段正文
    m2 = re.search(r'<div class="detail-text"[^>]*>(.*?)</div>\s*(?:<!--|</wbr>|<div[^>]*id="div_div")', html, re.S)
    if m2:
        return m2.group(1).strip()
    return ""

def extract_text(html_content):
    if not html_content:
        return ""
    text = re.sub(r'<[^>]+>', ' ', html_content)
    text = re.sub(r'\s+', ' ', text).strip()
    return text[:200]

def crawl():
    now = datetime.now()
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = c.fetchone()[0]
    incremental = existing > 0
    max_pages = 1 if incremental else MAX_PAGES
    total_new = 0
    total_skip = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = f"{BASE_URL}/default.htm"
        else:
            list_url = f"{BASE_URL}/default_{page-1}.htm"

        print(f"正在爬取第{page}页: {list_url}", flush=True)
        html = fetch(list_url)
        if not html:
            continue

        # 解析列表
        # 格式: <li><a href='content_123.html' target='_blank' title='...'>标题</a><span>2026-05-09</span></li>
        items = re.findall(
            r"<li>.*?<a href='(content_\d+\.html)'[^>]*title='([^']*)'[^>]*>(.*?)</a><span>([^<]+)</span>",
            html, re.DOTALL
        )

        if not items:
            print(f"第{page}页无数据", flush=True)
            break

        for href, title_attr, link_text, pub_date in items:
            title = title_attr.strip() or link_text.strip()
            if not title:
                continue
            pub_date = pub_date.strip()
            detail_url = f"{BASE_URL}/{href}"

            # 检查重复
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
            if c.fetchone():
                total_skip += 1
                continue

            # 获取详情
            detail_html = fetch(detail_url)
            if not detail_html:
                content = ""
                summary = title[:200]
            else:
                content = extract_content(detail_html)
                summary = extract_text(content)

            try:
                c.execute(
                    "INSERT INTO gov_raw (site_name, page_url, title, publish_date, summary, content, category, source_url) VALUES (?,?,?,?,?,?,?,?)",
                    (SITE_NAME, detail_url, title, pub_date, summary or title[:200], content, '环评公示',
                     "https://www.yiyang.gov.cn/yyshjbhj/3452/3467/default.htm")
                )
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_skip += 1
            except Exception as e:
                print(f"入库失败 [{title[:30]}]: {e}", flush=True)
                conn.rollback()

        print(f"第{page}页完成，本页{len(items)}条，新增累计{total_new}条，跳过{total_skip}条", flush=True)

    conn.close()
    print(f"\n益阳生态环境局环评爬取完成，共新增 {total_new} 条，跳过 {total_skip} 条", flush=True)

if __name__ == '__main__':
    crawl()
