#!/usr/bin/env python3
"""贵州省生态环境厅 - 建设项目受理公示 爬虫"""
import re, urllib.request, sys
from datetime import datetime

DB_PATH = "/root/search.db"
BASE_URL = "https://sthj.guizhou.gov.cn/zwgk/zdlyxx/hjpj/jsxmslgs"
SITE_NAME = "gzhb_jsxmslgs"
MAX_PAGES = 5  # 首次最多5页，日跑仅1页

def fetch(url, encoding='utf-8'):
    req = urllib.request.Request(url, headers={'User-Agent': 'Mozilla/5.0'})
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        return resp.read().decode(encoding, errors='replace')
    except Exception as e:
        print(f"请求失败 {url}: {e}", flush=True)
        return None

def extract_content(html):
    """提取 div#Zoom 内的正文"""
    m = re.search(r'<div id="Zoom"[^>]*>(.*?)</div>\s*(?:<div class="qrcode|<div class="article)', html, re.S)
    if m:
        return m.group(1).strip()
    return ""

def extract_text(html_content):
    if not html_content:
        return ""
    text = re.sub(r'<[^>]+>', ' ', html_content)
    text = re.sub(r'\s+', ' ', text).strip()
    return text[:200]

def crawl():
    now = datetime.now()
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()

    c.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = c.fetchone()[0]
    incremental = existing > 0
    max_pages = 1 if incremental else MAX_PAGES
    total_new = 0
    total_skip = 0

    for page in range(max_pages):
        if page == 0:
            list_url = f"{BASE_URL}/index.html"
        else:
            list_url = f"{BASE_URL}/index_{page}.html"

        print(f"正在爬取第{page+1}页: {list_url}", flush=True)
        html = fetch(list_url)
        if not html:
            continue

        # 解析列表: <li><a title="..." href="URL">标题</a><span>日期</span></li>
        items = re.findall(
            r'<li><a\s+target="_blank"\s+title="([^"]*)"\s+href="([^"]+)"[^>]*>.*?</a><span>([^<]+)</span>',
            html, re.DOTALL
        )

        if not items:
            print(f"第{page+1}页无数据", flush=True)
            items = re.findall(
                r"<li><a\s+target=\"_blank\"\s+title=\"([^\"]*)\"\s+href=\"([^\"]+)\"[^>]*>.*?</a><span>([^<]+)</span>",
                html, re.DOTALL
            )

        if not items:
            # Try looser match
            items = re.findall(
                r'<li>.*?<a[^>]*title="([^"]*?)"[^>]*href="([^"]+?)"[^>]*>(.*?)</a>\s*<span>([^<]+)</span>',
                html, re.DOTALL
            )
            if items:
                items = [(t, h, d) for t, h, _, d in items]
            else:
                continue

        print(f"本页{len(items)}条", flush=True)

        for title, href, pub_date in items:
            title = title.strip()
            if not title:
                continue
            pub_date = pub_date.strip()

            # URL处理
            if href.startswith('http'):
                detail_url = href
            elif href.startswith('/'):
                detail_url = f"https://sthj.guizhou.gov.cn{href}"
            else:
                detail_url = f"{BASE_URL}/{href}"

            # 检查重复
            c.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
            if c.fetchone():
                total_skip += 1
                continue

            # 获取详情
            detail_html = fetch(detail_url)
            if not detail_html:
                content = ""
                summary = title[:200]
            else:
                content = extract_content(detail_html)
                summary = extract_text(content)

            try:
                c.execute(
                    "INSERT INTO gov_raw (site_name, page_url, title, publish_date, summary, content, category, source_url) VALUES (?,?,?,?,?,?,?,?)",
                    (SITE_NAME, detail_url, title, pub_date, summary or title[:200], content, '建设项目受理公示',
                     "https://sthj.guizhou.gov.cn/zwgk/zdlyxx/hjpj/jsxmslgs/index.html")
                )
                conn.commit()
                total_new += 1
            except sqlite3.IntegrityError:
                total_skip += 1
            except Exception as e:
                print(f"入库失败 [{title[:30]}]: {e}", flush=True)
                conn.rollback()

        print(f"第{page+1}页完成，新增累计{total_new}条，跳过{total_skip}条", flush=True)

    conn.close()
    print(f"\n贵州省生态环境厅建设项目受理公示爬取完成，共新增 {total_new} 条，跳过 {total_skip} 条", flush=True)

if __name__ == '__main__':
    crawl()
