#!/usr/bin/env python3
"""
广东省生态环境厅 - 建设项目环评审批前公示 (gdee.gd.gov.cn)
https://gdee.gd.gov.cn/spqgs3191/
CMS: 省政府网站群, 静态分页 index_N.html
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests
import urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "广东省生态环境厅-环评审批公示"
BASE_URL = "https://gdee.gd.gov.cn"
LIST_PATH = "/spqgs3191"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except: return None

def parse_list(html):
    """解析列表页"""
    items = []
    # <li><a href="https://gdee.gd.gov.cn/spqgs3191/content/post_4905187.html">title</a></li>
    pattern = r'<a[^>]*href="[^\"]*/spqgs3191/content/post_(\d+)\.html"[^>]*>(.*?)</a>'
    for m in re.finditer(pattern, html, re.DOTALL):
        post_id = m.group(1)
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        full_url = f"{BASE_URL}/{LIST_PATH}/content/post_{post_id}.html"
        items.append((title, full_url))
    return items

def extract_nested_content(html, tag_start):
    """从tag_start位置开始，找到匹配的闭合标签，支持嵌套div/section"""
    depth = 0
    i = tag_start
    n = len(html)
    while i < n:
        # Skip HTML comments
        if html[i:i+4] == '<!--':
            end = html.find('-->', i+4)
            if end > i: i = end + 3
            else: i += 1
            continue
        # Check closing tags
        if html[i:i+6] == '</div>' or html[i:i+8] == '</section>':
            if depth == 0:
                return html[tag_start:i]
            depth -= 1
            i += 6 if html[i:i+6] == '</div>' else 8
            continue
        # Check opening tags
        ch = html[i]
        if ch == '<':
            next_chars = html[i+1:i+6]
            if next_chars in ('div', 'div '):
                depth += 1
                i += 4
                continue
            if next_chars[:4] in ('sect',) and html[i+1:i+8] == 'section':
                depth += 1
                i += 8
                continue
        i += 1
    return html[tag_start:]  # fallback

def fetch_detail(url):
    """获取详情页标题、正文、日期"""
    html = fetch(url)
    if not html: return None, None, None
    result = {}
    # 标题: <h1>xxx</h1> 或 <div class="headline ..."><p>xxx</p></div>
    m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m: result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    else:
        m = re.search(r'<div[^>]*class="headline[^"]*"[^>]*>\s*<p>(.*?)</p>', html, re.DOTALL)
        if m: result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    # 日期: <meta name="PubDate" content="2026-06-01"> 或 <div class="source ...">2026-06-01</div>
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="(\d{4}-\d{1,2}-\d{1,2})', html)
    if m: result["publish_date"] = m.group(1)
    else:
        m = re.search(r'<div[^>]*class="source[^"]*"[^>]*>\s*(\d{4}-\d{1,2}-\d{1,2})', html)
        if m: result["publish_date"] = m.group(1)
    # 正文: 使用嵌套感知提取
    for cls in ['article', 'content', 'TRS_Editor', 'text']:
        m = re.search(r'<(div|section)[^>]*class="' + cls + r'[^\"]*"[^>]*>', html, re.DOTALL)
        if m:
            content = extract_nested_content(html, m.end())
            content = content.strip()
            if len(content) > 200:
                content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL|re.I)
                content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL|re.I)
                result["content"] = content
                break
    return result.get("title"), result.get("content"), result.get("publish_date")

def main():
    print(f"\n{'='*50}")
    print(f"🏠 {SITE_NAME}")
    print(f"{'='*50}")
    all_list_items = []
    for page in range(1, MAX_PAGES+1):
        if page == 1:
            url = f"{BASE_URL}{LIST_PATH}/"
        else:
            url = f"{BASE_URL}{LIST_PATH}/index_{page}.html"
        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch(url)
        if not html: print("❌ 无返回"); break
        items = parse_list(html)
        if not items: print("0 条"); break
        print(f"✅ {len(items)} 条")
        all_list_items.extend(items)
    print(f"\n📊 列表总计: {len(all_list_items)} 条")
    all_items = []
    seen_urls = set()
    for i, (title, url) in enumerate(all_list_items):
        if "项目" not in title: continue
        if url in seen_urls: continue
        seen_urls.add(url)
        print(f"  [{i+1}/{len(all_list_items)}] {title[:50]}...", end=" ", flush=True)
        detail_title, content, detail_date = fetch_detail(url)
        if detail_title: title = detail_title
        summary = re.sub(r"<[^>]+>", " ", content or "").strip()
        summary = re.sub(r"\s+", " ", summary)[:300]
        all_items.append({
            "site_name": SITE_NAME, "title": title, "url": url,
            "content": content or "", "pub_date": detail_date or "",
            "summary": summary, "tags": SITE_NAME,
        })
        print(f"✅")
        time.sleep(0.3)
    if not all_items: print("⏭️ 无数据"); return
    push_to_searchdb(all_items, "gdee_spqgs")
    print(f"\n✅ 完成! 共 {len(all_items)} 条")

if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
