#!/usr/bin/env python3
"""
隆昌市 - 公示公告 (longchang.gov.cn)
https://www.longchang.gov.cn/lcs/gsgg/common_list.shtml
UCAP CMS, 详情: /lcs/gsgg/202606/uuid.shtml
内容通过 detaildata > data-article="content" 动态注入
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "隆昌市公示公告"
BASE_URL = "https://www.longchang.gov.cn"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except: return None

def parse_list(html):
    items = []
    pattern = r'href="(/lcs/gsgg/\d{6}/[^"]+\.shtml)"[^>]*>(.*?)</a>'
    for m in re.finditer(pattern, html, re.DOTALL):
        href = m.group(1)
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        if not title or len(title) < 5: continue
        items.append((title, BASE_URL + href))
    return items

def get_page_url(page):
    if page == 1: return f"{BASE_URL}/lcs/gsgg/common_list.shtml"
    return f"{BASE_URL}/lcs/gsgg/common_list_{page}.shtml"

def extract_detaildata_content(html):
    """从 detaildata 隐藏容器提取 data-article 字段"""
    dd = re.search(r'<div[^>]*class="detaildata[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if not dd:
        return {}
    result = {}
    inner = dd.group(1)
    # data-title
    m = re.search(r'data-article="data-title"[^>]*>\s*(.*?)\s*</div>', inner, re.DOTALL)
    if m: result['title'] = m.group(1).strip()
    # subTitle (fallback title)
    if not result.get('title'):
        m = re.search(r'data-article="subTitle"[^>]*>\s*(.*?)\s*</div>', inner, re.DOTALL)
        if m: result['title'] = m.group(1).strip()
    # publishedTime
    m = re.search(r'data-article="publishedTime"[^>]*>\s*(\d{4}-\d{1,2}-\d{1,2})', inner)
    if m: result['publish_date'] = m.group(1)
    # content
    m = re.search(r'data-article="content"[^>]*>(.*)', inner, re.DOTALL)
    if m:
        content = m.group(1)
        # 去除尾部多余的 </div> 闭合标签
        content = re.sub(r'</div>\s*$', '', content.strip())
        content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL|re.I)
        content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL|re.I)
        if len(content) > 100:
            result['content'] = content
    return result

def fetch_detail(url):
    html = fetch(url)
    if not html: return None, None, None
    result = {}
    # 尝试从 detaildata 提取
    dd = extract_detaildata_content(html)
    if dd.get('content'):
        result['content'] = dd['content']
        if dd.get('title'): result['title'] = dd['title']
        if dd.get('publish_date'): result['publish_date'] = dd['publish_date']
    # 标题兜底: <h1>
    if not result.get('title'):
        m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
        if m: result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    # 日期兜底: <meta PubDate>
    if not result.get('publish_date'):
        m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="(\d{4}-\d{1,2}-\d{1,2})', html)
        if m: result["publish_date"] = m.group(1)
    # 正文兜底：常见容器
    if not result.get('content'):
        for cls in ['content', 'article', 'text', 'main', 'wzcon', 'TRS_Editor']:
            m = re.search(r'<(div|section)[^>]*class="' + cls + r'[^"]*"[^>]*>(.*?)</\1>', html, re.DOTALL)
            if m and len(m.group(2)) > 200:
                content = m.group(2).strip()
                content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL|re.I)
                result["content"] = content
                break
    # 正文兜底：body
    if not result.get('content'):
        body = re.search(r'<body[^>]*>(.*?)</body>', html, re.DOTALL)
        if body:
            content = body.group(1)
            content = re.sub(r"<script[^>]*>.*?</script>|<style[^>]*>.*?</style>", "", content, flags=re.DOTALL|re.I)
            navs = re.findall(r'<(?:header|nav)[^>]*>.*?</(?:header|nav)>', content, re.DOTALL|re.I)
            for n in navs: content = content.replace(n, '')
            if len(content) > 100:
                result["content"] = content
    return result.get("title"), result.get("content"), result.get("publish_date")

def main():
    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")
    all_list = []
    for page in range(1, MAX_PAGES+1):
        url = get_page_url(page)
        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        html = fetch(url)
        if not html: print("❌"); break
        items = parse_list(html)
        if not items: print("0 条"); break
        print(f"✅ {len(items)} 条")
        all_list.extend(items)
    print(f"\n📊 列表总计: {len(all_list)} 条")
    all_items, seen = [], set()
    for i, (title, url) in enumerate(all_list):
        if "项目" not in title: continue
        if url in seen: continue
        seen.add(url)
        print(f"  [{i+1}/{len(all_list)}] {title[:50]}...", end=" ", flush=True)
        dt, content, date = fetch_detail(url)
        if dt: title = dt
        summary = re.sub(r"<[^>]+>", " ", content or "").strip()[:300]
        summary = re.sub(r"\s+", " ", summary)
        all_items.append({
            "site_name": SITE_NAME, "title": title, "url": url,
            "content": content or "", "pub_date": date or "",
            "summary": summary, "tags": SITE_NAME,
        })
        print("✅")
        time.sleep(0.3)
    if all_items:
        push_to_searchdb(all_items, "longchang_gsgg")
    print(f"\n✅ 完成! 共 {len(all_items)} 条")

if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
