#!/usr/bin/env python3
"""
大同市生态环境局-公示栏 (sthjj.dt.gov.cn/dtssthjjz/gsl/listN.shtml)
UCAP CMS: <ul class="list-cj-gl"> + data-title/data-url/data-time
详情: <div class='TRS_Editor'><ucapcontent> 正文
单页列表（无分页）
"""
import sys, os, re, time, html as html_mod
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "大同市生态环境局-公示栏"
BASE_URL = "https://sthjj.dt.gov.cn"
LIST_URL = BASE_URL + "/dtssthjjz/gsl/listN.shtml"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch(url):
    for _ in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=20)
            r.encoding = "utf-8"
            if r.text.strip():
                return r.text
        except:
            pass
        time.sleep(2)
    return None

def parse_list(html):
    """提取 (title, url_path, date_str) 从 data-* 属性"""
    items = []
    for m in re.finditer(
        r'<li data-title="([^"]*)" data-url="([^"]*)" data-time="([^"]*)">',
        html
    ):
        title = html_mod.unescape(m.group(1)).strip()
        url_path = m.group(2).strip()
        date_str = m.group(3).strip()
        if title and url_path:
            items.append((title, url_path, date_str))
    return items

def fetch_content(url):
    """从详情页提取 <div class='TRS_Editor'> 下 <ucapcontent> 正文"""
    html = fetch(url)
    if not html:
        return ""

    # UCAP CMS: <div class='TRS_Editor'><ucapcontent>...</ucapcontent></div>
    m = re.search(r"<div[^>]*class=['\"]TRS_Editor['\"][^>]*>(.*?)</div>", html, re.DOTALL)
    if m:
        editor_content = m.group(1)
        # 提取 ucapcontent 内容
        uc = re.search(r"<ucapcontent>(.*?)</ucapcontent>", editor_content, re.DOTALL)
        if uc:
            content = uc.group(1).strip()
            if content:
                # 修正相对路径的图片/附件
                content = re.sub(r'(src|href)="(?!http)(?!javascript)([^"]+)"',
                    lambda m: f'{m.group(1)}="{url.rsplit("/", 1)[0]}/{m.group(2)}"' if not m.group(2).startswith(('#', '/')) else m.group(0),
                    content)
                return content
        # fallback: TRS_Editor 内全部内容
        raw = editor_content.strip()
        raw = re.sub(r"<ucapcontent>|</ucapcontent>", "", raw).strip()
        if raw:
            return raw

    return ""

def main(incremental=False, limit=None):
    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")

    html = fetch(LIST_URL)
    if not html:
        print("❌ 列表页获取失败")
        return

    items = parse_list(html)
    if not items:
        print("❌ 未找到条目")
        return

    print(f"📊 共 {len(items)} 条")

    all_items, seen = [], set()
    for title, url_path, list_date in items:
        if list_date and list_date < THREE_YEARS_AGO:
            continue

        # 构造完整 URL
        if url_path.startswith('http'):
            full_url = url_path
        else:
            # ../../dtssthjjz/gsl/202606/xxx.shtml
            clean = url_path.lstrip('./')
            full_url = f"{BASE_URL}/{clean}"

        if full_url in seen:
            continue
        seen.add(full_url)

        print(f"  [{len(all_items)+1}] {list_date} {title[:45]}...", end=" ", flush=True)
        content = fetch_content(full_url)

        summary = re.sub(r"<[^>]+>", " ", content).strip()[:300]
        summary = re.sub(r"\s+", " ", summary) if summary else title
        print(f"✅ ({len(content)} chars)")

        all_items.append({
            "site_name": SITE_NAME,
            "title": title,
            "url": full_url,
            "content": content,
            "pub_date": list_date,
            "summary": summary,
            "tags": SITE_NAME,
        })

        time.sleep(0.3)
        if limit and len(all_items) >= limit:
            break

    if all_items:
        push_to_searchdb(all_items, "datong_gsl")
        print(f"\n✅ 完成! 共 {len(all_items)} 条")
    else:
        print("\n⏭ 无新数据")

if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--incremental", action="store_true")
    parser.add_argument("--limit", type=int)
    args = parser.parse_args()
    incremental = args.incremental or (len(sys.argv) > 1 and sys.argv[1] == "1")
    t0 = time.time()
    main(incremental=incremental, limit=args.limit)
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
