#!/usr/bin/env python3
"""谷城县生态环境局 - 回应关切(环评公示) 爬虫"""
import re, urllib.request, sys
from datetime import datetime

DB_PATH = "/root/search.db"
BASE_URL = "http://www.hbgucheng.gov.cn/bmxz/bm/hbj/fdzdgk2/qtzdgknr/hygq_32016"
SITE_NAME = "gucheng_hbj"
MAX_PAGES = 3  # 总共3页

UA = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36'

def fetch(url, encoding='utf-8'):
    req = urllib.request.Request(url, headers={'User-Agent': UA})
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        return resp.read().decode(encoding, errors='replace')
    except Exception as e:
        print(f"请求失败 {url}: {e}", flush=True)
        return None

def extract_content(html):
    """提取 div#cbcontent 内的正文"""
    m = re.search(r'<div id="cbcontent"[^>]*>(.*?)</div>\s*</div>', html, re.S)
    if m:
        return m.group(1).strip()
    return ""

def extract_text(html_content):
    if not html_content:
        return ""
    text = re.sub(r'<[^>]+>', ' ', html_content)
    text = re.sub(r'\s+', ' ', text).strip()
    return text[:200]

def crawl():
    now = datetime.now()
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = c.fetchone()[0]
    incremental = existing > 0
    max_pages = 1 if incremental else MAX_PAGES
    total_new = 0
    total_skip = 0

    for page in range(max_pages):
        if page == 0:
            list_url = f"{BASE_URL}/index.shtml"
        else:
            list_url = f"{BASE_URL}/index_{page}.shtml"

        print(f"正在爬取第{page+1}页: {list_url}", flush=True)
        html = fetch(list_url)
        if not html:
            continue

        items = re.findall(
            r'<li>.*?<a href="([^"]+)"[^>]*title="([^"]*)"[^>]*>\s*(.*?)\s*<span>([^<]+)</span>',
            html, re.DOTALL
        )

        if not items:
            print(f"第{page+1}页无数据", flush=True)
            continue

        print(f"本页{len(items)}条", flush=True)

        for href, title_attr, link_text, pub_date in items:
            title = title_attr.strip() or re.sub(r'<[^>]+>', '', link_text).strip()
            if not title:
                continue
            pub_date = pub_date.strip()

            if href.startswith('http'):
                detail_url = href
            elif href.startswith('/'):
                detail_url = f"http://www.hbgucheng.gov.cn{href}"
            else:
                detail_url = f"{BASE_URL}/{href}"

            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
            if c.fetchone():
                total_skip += 1
                continue

            detail_html = fetch(detail_url)
            if not detail_html:
                content = ""
                summary = title[:200]
            else:
                content = extract_content(detail_html)
                summary = extract_text(content)

            try:
                c.execute(
                    "INSERT INTO gov_raw (site_name, page_url, title, publish_date, summary, content, category, source_url) VALUES (?,?,?,?,?,?,?,?)",
                    (SITE_NAME, detail_url, title, pub_date, summary or title[:200], content, '环评公示',
                     "http://www.hbgucheng.gov.cn/bmxz/bm/hbj/fdzdgk2/qtzdgknr/hygq_32016/index.shtml")
                )
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_skip += 1
            except Exception as e:
                print(f"入库失败 [{title[:30]}]: {e}", flush=True)
                conn.rollback()

        print(f"第{page+1}页完成，新增累计{total_new}条，跳过{total_skip}条", flush=True)

    conn.close()
    print(f"\n谷城县生态环境局环评公示爬取完成，共新增 {total_new} 条，跳过 {total_skip} 条", flush=True)

if __name__ == '__main__':
    crawl()
