#!/usr/bin/env python3
"""
crawl_tongcheng_hpgs.py - 桐城市生态环境分局 (www.tongcheng.gov.cn)
站点: CDN 521 状态但内容正常; playwright 过盾
列表: GET /site/label/8888 (注意: 无 huangshanzwgk 前缀, 与 ahshx 不同)
  labelName=publicInfoList&siteId=2000004241&organId={organ}&pageSize=10&pageIndex=N
  &isDate=true&dateFormat=yyyy-MM-dd&length=50&type=4&action=list&isJson=true&catId={cat}
详情: /public/{organ}/{contentId}.html, meta ArticleTitle/PubDate, 正文 div.wzcon.j-fontContent

栏目 (--col):
  hpgs : 桐城市生态环境分局-环评公示/意见征集 organ=2000002271 cat=6725591 (total=75)
  luting_yjzj : 桐城市吕亭镇-意见征集 organ=2000002421 cat=6736061 (total=14, 含规划公示/环评公示/征求意见)

进程锁: flock 站点级串行化
"""
import sys, os, re, asyncio, json, fcntl, time
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE = "https://www.tongcheng.gov.cn"
SITE_ID = "2000004241"
LOCK_FILE = "/tmp/tongcheng_hpgs.lock"

COL_MAP = {
    "hpgs": {"organ": "2000002271", "cat": "6725591", "name": "桐城市生态环境分局-环评公示"},
    "luting_yjzj": {"organ": "2000002421", "cat": "6736061", "name": "桐城市吕亭镇-意见征集"},
}

UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"


def clean_html(html):
    """白名单清洗: 去 inline style, unwrap span/font 装饰, 保留 p/table/a, 链接绝对化"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    for t in soup(['script', 'style', 'iframe', 'object']):
        t.decompose()
    for img in soup.find_all('img'):
        src = img.get('src') or img.get('data-src') or ''
        if src:
            abs_src = src if src.startswith('http') else BASE + src
            a = soup.new_tag('a', href=abs_src)
            a.string = '查看图片'
            img.replace_with(a)
        else:
            img.decompose()
    for t in soup.find_all(['span', 'font', 'b', 'strong', 'em', 'i', 'u', 's', 'div']):
        t.unwrap()
    for a in soup.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith('javascript'):
            a['href'] = href if href.startswith('http') else BASE + href
        for attr in list(a.attrs):
            if attr not in ('href',):
                del a[attr]
    for p in soup.find_all('p'):
        for attr in list(p.attrs):
            del p[attr]
        if not p.get_text(strip=True) and not p.find('table') and not p.find('img'):
            p.decompose()
    for t in soup.find_all(['table', 'tr', 'td', 'th']):
        for attr in list(t.attrs):
            if attr not in ('border', 'cellpadding', 'cellspacing'):
                del t[attr]
    # 清理 PDF 外链站的固定提示尾巴 (纯文本节点 + 链接另存为 a 标签)
    for node in list(soup.strings):
        txt = node.strip()
        if txt and ('如果内容不能正常显示' in txt or '下载本PDF文档' in txt or 'pdf浏览器插件' in txt):
            parent = node.parent
            if parent and parent.name in ('p', 'div'):
                parent.decompose()
            else:
                node.extract()
    for a in soup.find_all('a'):
        if '链接另存为' in a.get_text() or '鼠标右键点击' in a.get_text():
            a.decompose()
    # 清理孤立尾部文本节点: 仅剩 ']。' 之类 PDF 尾巴残渣
    for node in list(soup.strings):
        txt = node.strip()
        if txt and len(txt) <= 5 and txt in (']。', '。]', '】。', '】', '。', '）', ')', ']', '】'):
            node.extract()
    return str(soup)


def acquire_lock(max_wait=1800):
    """站点级互斥锁, 阻塞等待最多 max_wait 秒"""
    fd = open(LOCK_FILE, 'w')
    try:
        fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
        return fd
    except BlockingIOError:
        print(f"  [LOCK] 另一 ahshx 实例运行中, 等待释放 (最多 {max_wait}s)...")
        deadline = time.time() + max_wait
        while time.time() < deadline:
            try:
                fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
                return fd
            except BlockingIOError:
                time.sleep(10)
        print("  [LOCK] 等待超时, 放弃")
        return None


async def run_async(col_key, pages=1):
    col = COL_MAP[col_key]
    organ_id = col["organ"]
    cat_id = col["cat"]
    site_name = col["name"]

    API_TPL = (f"{BASE}/site/label/8888?_=0.{{rand}}"
               f"&labelName=publicInfoList&siteId={SITE_ID}&organId={organ_id}"
               f"&pageSize=10&pageIndex={{page}}&isDate=true&dateFormat=yyyy-MM-dd&length=50"
               f"&type=4&action=list&result=&isJson=true&keyWords=&isSetValue=true&catIds=&catId={cat_id}")
    LIST_URL = f"{BASE}/public/column/{organ_id}?type=4&catId={cat_id}&action=list&nav=3"

    from playwright.async_api import async_playwright
    from playwright_stealth import Stealth
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path='/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome',
            args=['--no-sandbox', '--disable-setuid-sandbox', '--disable-blink-features=AutomationControlled']
        )
        ctx = await browser.new_context(user_agent=UA, locale='zh-CN',
                                        viewport={'width': 1440, 'height': 900}, timezone_id='Asia/Shanghai')
        stealth = Stealth()
        await stealth.apply_stealth_async(ctx)
        page = await ctx.new_page()

        # 过盾: 访问列表页
        try:
            resp = await page.goto(LIST_URL, wait_until='domcontentloaded', timeout=35000)
            await page.wait_for_timeout(12000)
            html = await page.content()
            m = re.search(r'<title>([^<]*)</title>', html)
            print(f"  [LOAD] status={resp.status if resp else '?'} title={m.group(1) if m else '?'} len={len(html)}")
            if m and ('NWAF' in m.group(1) or 'Environment' in m.group(1)):
                print("  [WAF] 仍被拦截")
                await browser.close()
                return 0
        except Exception as e:
            print(f"  [LOAD ERR] {str(e)[:120]}")
            await browser.close()
            return 0

        # API 拉列表
        items_all = []
        seen = set()
        total = 0
        for pg in range(1, pages + 1):
            api_url = API_TPL.format(rand=str(abs(hash(f"tongcheng_{col_key}_{pg}")))[-15:], page=pg)
            data = await page.evaluate(f"""async () => {{
                try {{
                    const r = await fetch('{api_url}');
                    return await r.json();
                }} catch(e) {{ return {{error: e.message}}; }}
            }}""")
            if 'error' in data:
                print(f"  [API ERR] {data['error']}")
                break
            total = data.get('total', 0)
            items = data.get('data', []) or []
            if not items:
                break
            print(f"  第{pg}页: {len(items)} 条 (total={total})")
            for it in items:
                link = it.get('link', '')
                if link and link not in seen:
                    seen.add(link)
                    items_all.append({
                        'url': link if link.startswith('http') else BASE + link,
                        'title': it.get('title', '').strip(),
                        'date': (it.get('publishDate') or '')[:10],
                    })
            if len(items_all) >= total:
                break

        print(f"  [LIST] total={total}, 去重后 {len(items_all)} 条")

        # 详情页
        records = []
        for idx, it in enumerate(items_all):
            print(f"  [{idx+1}/{len(items_all)}] {it['title'][:40]}...")
            try:
                resp = await page.goto(it['url'], wait_until='domcontentloaded', timeout=30000)
                await page.wait_for_timeout(3000)
                html = await page.content()
            except Exception as e:
                print(f"    [DETAIL ERR] {str(e)[:100]}")
                continue
            # meta
            title = it['title']
            m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
            if m and m.group(1).strip():
                title = m.group(1).strip()
            pub_date = it['date']
            m = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html)
            if m:
                dm = re.search(r'\d{4}-\d{2}-\d{2}', m.group(1))
                if dm:
                    pub_date = dm.group()
            # 正文 wzcon j-fontContent (DOM 提取最干净, 避免正则吞入二维码/按钮)
            content = ""
            try:
                content = await page.evaluate("document.querySelector('.j-fontContent') ? document.querySelector('.j-fontContent').innerHTML : ''")
            except Exception:
                pass
            if not content.strip():
                m = re.search(r'class="wzcon\s+j-fontContent[^"]*"[^>]*>([\s\S]*?)</div>\s*</div>', html)
                if m:
                    content = m.group(1).strip()
            if not content.strip():
                print("    [SKIP] 正文空")
                continue
            # 清洗: 去 inline style / unwrap span/font / 链接绝对化
            content = clean_html(content)
            records.append({
                "title": title,
                "url": it['url'],
                "pub_date": pub_date,
                "site_name": site_name,
                "content": content,
                "summary": "",
            })
        await browser.close()

    print(f"  共 {len(records)} 条有效")
    if records:
        push_to_searchdb(records, f"ahshx_{col_key}")
    return len(records)


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--col", default="xitou_zb", help="栏目键: " + ",".join(COL_MAP.keys()))
    parser.add_argument("--pages", type=int, default=1)
    parser.add_argument("--all", action="store_true")
    args = parser.parse_args()
    if args.col not in COL_MAP:
        print(f"[ERR] 未知栏目 {args.col}, 可选: {list(COL_MAP.keys())}")
        return 1

    lock_fd = acquire_lock()
    if lock_fd is None:
        return 1
    try:
        col = COL_MAP[args.col]
        print(f"  {col['name']} (organ={col['organ']} cat={col['cat']}) pages={args.pages}")
        cnt = asyncio.run(run_async(args.col, pages=args.pages if not args.all else 50))
        print(f"Done: {cnt} records")
        return 0
    finally:
        fcntl.flock(lock_fd, fcntl.LOCK_UN)
        lock_fd.close()


if __name__ == "__main__":
    sys.exit(main())
