#!/usr/bin/env python3
"""crawl_lianxi_tzgg.py - 濂溪区人民政府-政务资讯-通知公告 (MS UI 动态列表)

站点: https://www.lianxi.gov.cn/zwzx/tzgg/
CMS: TRS 详情 (zwdetail_3_1 TRS_UEDITOR div#content) + MS UI (ms-controller) 动态年份列表
列表: <li><span class="bf-pass">日期</span><a href="./YYYYMM/tYYYYMMDD_ID.html?dD1..." title="标题">
      年份筛选 ?year=YYYY&page=N (仅2026:4页61条/2025:2页23条, 越界页回退默认非空!)
      混合外链 api.phxazx.cn (濂溪新闻网) → 过滤只留本域 /zwzx/tzgg/\d{6}/t\d+_\d+\.html
详情: meta[ArticleTitle] + meta[PubDate] + div#content 平衡div (requests 直抓, 无需playwright)
附件: ./P0...docx/pdf 相对路径 → urljoin 绝对化
样本: 濂溪区2025年省级财政支持第二轮土地承包到期后再延长30年试点项目实施方案 (2026-08-07)
"""
import sys, os, re, argparse, time
from datetime import datetime, timedelta
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = (datetime.now() - timedelta(days=3 * 365)).strftime("%Y-%m-%d")
SITE_NAME = '濂溪区人民政府-通知公告'
GROUP_NAME = '江西'
BASE = 'https://www.lianxi.gov.cn'
LIST_URL = BASE + '/zwzx/tzgg/'
YEARS = [2026, 2025]
PAGES_PER_YEAR = 4
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Referer': LIST_URL,
}


def fetch_list_with_playwright(year, page_num):
    """MS UI 动态列表必须 playwright 渲染"""
    from playwright.sync_api import sync_playwright
    url = f'{LIST_URL}?year={year}&page={page_num}'
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True)
        ctx = browser.new_context(
            user_agent='Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
            locale='zh-CN')
        page = ctx.new_page()
        page.goto(url, timeout=60000, wait_until='networkidle')
        page.wait_for_timeout(2500)
        html = page.content()
        browser.close()
    items = re.findall(r'<li><span class="bf-pass">([\d-]+)</span><a href="([^"]+)"[^>]*title="([^"]*)"', html)
    results = []
    for d, href, title in items:
        title = re.sub(r'\s+', '', title).strip()
        if len(title) < 4:
            continue
        # 只留本域 tzgg 文章 (过滤 api.phxazx.cn 外链 + 去查询参数)
        m = re.match(r'\.?/?(\d{6}/t\d+_\d+\.html)', href)
        if not m:
            continue
        full = urljoin(LIST_URL, href.split('?')[0])
        results.append({'url': full, 'pub_date': d, 'title': title})
    return results


def fetch_detail(url):
    import requests
    requests.packages.urllib3.disable_warnings()
    try:
        r = requests.get(url, timeout=(8, 20), headers=HEADERS, verify=False)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        print(f'[WARN] fetch failed: {url} - {e}', file=sys.stderr)
        return None
    m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    title = m.group(1).strip() if m else ''
    m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
    pub_date = m.group(1)[:10] if m else ''
    # 正文: div#content 平衡 div
    m = re.search(r'<div[^>]*id="content"[^>]*>', html, re.I)
    if not m:
        return None
    gt = html.find('>', m.start())
    i = gt + 1
    depth = 1
    body = ''
    for mm in re.finditer(r'<div[\s>]|</div>', html[i:], re.I):
        if mm.group(0).startswith('<div'):
            depth += 1
        else:
            depth -= 1
            if depth == 0:
                body = html[i:i + mm.start()]
                break
    if not body.strip():
        return None
    body = re.sub(r'<script.*?</script>', '', body, flags=re.DOTALL | re.I)
    body = re.sub(r'<style.*?</style>', '', body, flags=re.DOTALL | re.I)
    # 规范 <p style=...> 为 <p> (search_app 检测 '<p>' 严格闭合才走 HTML 渲染)
    body = re.sub(r'<p[^>]*>', '<p>', body)
    body = re.sub(r'href="(?!https?://|//|#|javascript:)([^"]+)"',
                  lambda mm: f'href="{urljoin(url, mm.group(1))}"', body)
    body = re.sub(r'src="(?!https?://|//)([^"]+)"',
                  lambda mm: f'src="{urljoin(url, mm.group(1))}"', body)
    body = re.sub(r'^(<p[^>]*>(\s*(<br\s*/?>)?\s*)?</p>)+', '', body)
    body = re.sub(r'(<p[^>]*>(\s*(<br\s*/?>)?\s*)?</p>)+$', '', body)
    body = body.strip()
    if not body:
        return None
    return {'title': title, 'pub_date': pub_date, 'content': body}


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument('--pages', type=int, default=PAGES_PER_YEAR, help='每年份页数')
    ap.add_argument('--max-detail', type=int, default=0)
    args = ap.parse_args()

    print(f'=== {SITE_NAME} === CUTOFF: {CUTOFF}')
    # Step 1: 列表 (playwright 渲染)
    all_recs = []
    seen_url = set()
    for year in YEARS:
        for pg in range(1, args.pages + 1):
            items = fetch_list_with_playwright(year, pg)
            # 越界页回退默认(2026 P1): 检测到全部 URL 已见过即停
            new_items = [it for it in items if it['url'] not in seen_url]
            print(f'  year={year} P{pg}: {len(items)} 条 (新增 {len(new_items)})')
            if pg > 1 and not new_items:
                break
            for it in items:
                if it['url'] in seen_url:
                    continue
                seen_url.add(it['url'])
                all_recs.append(it)
    print(f'列表去重合计: {len(all_recs)} 条')

    # Step 2: CUTOFF 过滤 + 详情
    records = []
    for i, it in enumerate(all_recs):
        if it['pub_date'] and it['pub_date'] < CUTOFF:
            continue
        d = fetch_detail(it['url'])
        if not d:
            continue
        records.append({
            'title': d['title'] or it['title'],
            'url': it['url'],
            'source_url': it['url'],
            'pub_date': d['pub_date'] or it['pub_date'],
            'site_name': SITE_NAME,
            'content': d['content'],
            'summary': '',
            'group_name': GROUP_NAME,
        })
        if args.max_detail and len(records) >= args.max_detail:
            break
        time.sleep(0.3)
    print(f'窗口内: {len(records)} 条')

    if records:
        push_to_searchdb(records, 'lianxi_tzgg')
    print(f'[RESULT] {SITE_NAME}: {len(records)} items')


if __name__ == '__main__':
    main()
