#!/usr/bin/env python3
"""
郑州航空港区 - 环评公示 (zzhkgq.gov.cn)
https://www.zzhkgq.gov.cn/zwxx/hpgs/
CMS: 政府网站群, 日期式URL
正文: content-txt, 附件: content-relative attachment
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "郑州航空港区-环评公示"
BASE_URL = "https://www.zzhkgq.gov.cn"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except: return None

def parse_list(html):
    items = []
    pattern = r'<a[^>]*href="(https?://[^"]*zzhkgq[^"]*/\d{4}/\d{2}-\d{2}/\d+\.html)"[^>]*>(.*?)</a>'
    for m in re.finditer(pattern, html, re.DOTALL):
        href = m.group(1)
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        if len(title) < 5: continue
        items.append((title, href))
    return items

def extract_nested_content(html, tag_start):
    """找到匹配的闭合标签，支持嵌套div"""
    depth = 0
    i = tag_start
    n = len(html)
    while i < n:
        if html[i:i+4] == '<!--':
            end = html.find('-->', i+4)
            if end > i: i = end + 3
            else: i += 1
            continue
        if html[i:i+6] == '</div>':
            if depth == 0:
                return html[tag_start:i]
            depth -= 1
            i += 6
            continue
        if html[i:i+4] == '<div' and html[i+4] in (' ', '>', '\n', '\t', '\r'):
            depth += 1
            i += 4
            continue
        i += 1
    return html[tag_start:]

def fetch_detail(url):
    html = fetch(url)
    if not html: return None, None, None
    result = {}
    # 标题: <div class="content-title">xxx</div>
    m = re.search(r'<div[^>]*class="content-title"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m: result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    if not result.get("title"):
        m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
        if m: result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    # 日期: <meta PubDate> 或 <em>[2026-06-03]</em> 或 <input id="pubTime">
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="(\d{4}-\d{1,2}-\d{1,2})', html)
    if m: result["publish_date"] = m.group(1)
    if not result.get("publish_date"):
        m = re.search(r'<em[^>]*>\s*\[(\d{4}-\d{1,2}-\d{1,2})\]\s*</em>', html)
        if m: result["publish_date"] = m.group(1)
    if not result.get("publish_date"):
        m = re.search(r'id="pubTime"[^>]*value="\[?(\d{4}-\d{1,2}-\d{1,2})\]?"', html)
        if m: result["publish_date"] = m.group(1)
    # 正文: content-txt (嵌套感知)
    m = re.search(r'<div[^>]*class="content-txt[^"]*"[^>]*>', html)
    parts = []
    if m:
        content = extract_nested_content(html, m.end()).strip()
        if len(content) > 100:
            parts.append(content)
    # 附件: content-relative attachment
    m = re.search(r'<div[^>]*class="content-relative\s+attachment[^"]*"[^>]*>', html)
    if m:
        attachment = extract_nested_content(html, m.end()).strip()
        if len(attachment) > 10:
            parts.append(attachment)
    if parts:
        combined = "\n".join(parts)
        combined = re.sub(r"<script[^>]*>.*?</script>", "", combined, flags=re.DOTALL|re.I)
        combined = re.sub(r"<style[^>]*>.*?</style>", "", combined, flags=re.DOTALL|re.I)
        result["content"] = combined
    else:
        # 兜底：常见容器
        for cls in ['content', 'main', 'TRS_Editor', 'article', 'text']:
            m = re.search(r'<div[^>]*class="' + cls + r'[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
            if m and len(m.group(1)) > 100:
                content = m.group(1).strip()
                content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL|re.I)
                result["content"] = content
                break
    return result.get("title"), result.get("content"), result.get("publish_date")

def main():
    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")
    all_items = []
    seen_urls = set()
    for page in range(1, MAX_PAGES+1):
        url = f"{BASE_URL}/zwxx/hpgs/"
        if page > 1: url += f"index_{page}.html"
        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch(url)
        if not html: print("❌"); break
        items = parse_list(html)
        if not items: print("0 条"); break
        items = [it for it in items if it[0] not in seen_urls]
        print(f"✅ {len(items)} 条")
        for title, href in items:
            if "项目" not in title: continue
            if href in seen_urls: continue
            seen_urls.add(href)
            print(f"  {title[:50]}...", end=" ", flush=True)
            dt, content, date = fetch_detail(href)
            if dt: title = dt
            summary = re.sub(r"<[^>]+>", " ", content or "").strip()[:300]
            summary = re.sub(r"\s+", " ", summary)
            all_items.append({
                "site_name": SITE_NAME, "title": title, "url": href,
                "content": content or "", "pub_date": date or "",
                "summary": summary, "tags": SITE_NAME,
            })
            print("✅")
            time.sleep(0.3)
    if all_items:
        push_to_searchdb(all_items, "zzhkgq_hpgs")
    print(f"\n✅ 完成! 共 {len(all_items)} 条")

if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
