#!/usr/bin/env python3
"""爬虫：衢州市生态环境局公示公告（Playwright渲染）
www.qz.gov.cn/col/col1229039395/index.html?number=A09
注：列表JS渲染，需Playwright；详情页用requests
"""

import os, re, time, json, asyncio
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from playwright.async_api import async_playwright

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
LIST_URL = "https://www.qz.gov.cn/col/col1229039395/index.html?number=A09"
SITE_NAME = "衢州市生态环境局公示公告"
MAX_CLICKS = 80  # 1240条/15条每页≈83页

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

async def fetch_all_links():
    """用Playwright渲染页面，点击更多/翻页获取所有列表项"""
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True, args=["--no-sandbox"])
        page = await browser.new_page()
        await page.goto(LIST_URL, wait_until="commit", timeout=60000)
        await page.wait_for_timeout(8000)

        # 点一次"更多"切换为分页模式
        try:
            await page.evaluate('document.querySelector("span.more")?.click()')
            await page.wait_for_timeout(3000)
        except:
            pass

        all_items = {}
        # 翻页遍历
        for pg in range(1, MAX_CLICKS + 1):
            items = await page.evaluate("""
                () => {
                    const lis = document.querySelectorAll('ul.ajax-ul li.cf');
                    return Array.from(lis).map(li => {
                        const a = li.querySelector('a');
                        const span = li.querySelector('span.fr');
                        return {
                            title: a ? (a.getAttribute('title') || a.textContent.trim()) : '',
                            date: span ? span.textContent.trim() : '',
                            url: a ? a.href : ''
                        };
                    });
                }
            """)
            for item in items:
                if item["url"] and item["url"] not in all_items:
                    all_items[item["url"]] = item

            print(f"  翻页{pg}: {len(items)}条显示, 累计{len(all_items)}唯一URL")

            # 找下一页按钮（通过JS点击）
            try:
                has_next = await page.evaluate("""() => {
                    const n = document.querySelector('.layui-laypage-next');
                    return n && !n.classList.contains('layui-disabled');
                }""")
                if has_next:
                    await page.evaluate('document.querySelector(".layui-laypage-next")?.click()')
                    await page.wait_for_timeout(1500)
                else:
                    print("  无下一页，终止")
                    break
            except:
                print("  翻页失败，终止")
                break

        await browser.close()
        return list(all_items.values())

def extract_detail(detail_url):
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30, allow_redirects=True)
        r.encoding = "utf-8"
    except Exception:
        return None, None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")
    h2 = soup.select_one(".art_title h2")
    title = h2.get_text(strip=True) if h2 else ""

    pub_date = ""
    source_text = ""
    fz = soup.select_one(".fz_xx")
    if fz:
        text = fz.get_text(strip=True)
        m = re.search(r"发布[日期：:]\s*(\d{4}-\d{2}-\d{2})", text)
        if m:
            pub_date = m.group(1)
        m2 = re.search(r"信息[来源来源：:]\s*([^\s<]+)", text)
        if m2:
            source_text = m2.group(1).strip()

    zoom = soup.select_one("#zoom")
    if not zoom:
        return title, pub_date, source_text, "", ""

    attachments = []
    for a_tag in zoom.find_all("a"):
        href = a_tag.get("href", "")
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href.lower()) or "download" in href:
            text = a_tag.get_text(strip=True)
            full_url = urljoin(detail_url, href)
            attachments.append(f'<p><a href="{full_url}">{text}</a></p>')

    parts = []
    has_table = bool(zoom.find("table"))
    for tag in zoom.find_all(["p", "table"]):
        if tag.name == "p" and tag.find_parent("table"):
            continue
        if tag.name == "table":
            if has_table:
                parts.append(str(tag))
            continue
        text = tag.get_text(strip=True)
        text = re.sub(r"\n+", "", text)
        if text:
            parts.append(text)

    if not parts:
        text = re.sub(r"\n+", "", zoom.get_text(strip=True))
        if text:
            parts.append(text)

    content = "\n\n".join(parts)
    attachments_str = "\n".join(attachments) if attachments else ""

    if not content.strip() and not attachments:
        imgs = zoom.find_all("img")
        if imgs:
            img_parts = []
            for img in imgs:
                src = img.get("src", "")
                alt = img.get("alt", "")
                if src:
                    full_src = urljoin(detail_url, src)
                    img_parts.append(f'<p><a href="{full_src}">查看图片</a></p>')
            if img_parts:
                content = f'<p><a href="{detail_url}">{title}</a></p>\n\n' + "\n\n".join(img_parts)

    if not content.strip() and attachments:
        content = f'<p><a href="{detail_url}">{title}</a></p>\n\n{attachments_str}'

    return title, pub_date, source_text, content, attachments_str

def crawl():
    import sqlite3

    print("[PLAYWRIGHT] 加载页面获取列表...")
    try:
        all_items = asyncio.run(fetch_all_links())
    except Exception as e:
        print(f"[ERROR] Playwright失败: {e}")
        # 降级：直接调API取前5条
        print("[FALLBACK] 降级API模式取前5条")
        import requests as req
        api_url = "https://www.qz.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
        api_params = {
            "parseType": "bulidstatic", "webId": "3084",
            "tplSetId": "i58lqHVn2cokOgipAEjQF", "pageType": "column",
            "tagId": "组配分类list", "pageId": "1229039395",
        }
        r = req.get(api_url, params=api_params, headers=HEADERS, timeout=15)
        html = r.json()["data"]["html"]
        all_items = []
        for m in re.finditer(r'<li class="cf border-line">(.*?)</li>', html, re.DOTALL):
            li = m.group(1)
            a = re.search(r'href="([^"]+)"[^>]*title="([^"]*)"', li)
            span = re.search(r'<span class="fr">([^<]+)</span>', li)
            if a and span:
                all_items.append({"title": a.group(2), "date": span.group(1).strip(), "url": a.group(1)})

    print(f"共{len(all_items)}条列表项")

    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    total = 0

    for idx, item in enumerate(all_items):
        title = item["title"]
        list_date = item["date"]
        detail_url = item["url"]
        print(f"  [{idx+1}/{len(all_items)}] {title[:40]}...", end=" ", flush=True)

        full_title, pub_date, source_text, content, attachments_str = extract_detail(detail_url)
        if not full_title:
            full_title = title
        if not pub_date:
            pub_date = list_date

        date_rank = 0
        if pub_date:
            try:
                date_rank = 0 - int(pub_date.replace("-", "") + "0000")
            except ValueError:
                date_rank = 0

        try:
            cur.execute("""
                INSERT OR REPLACE INTO gov_raw
                (page_url, site_name, title, publish_date, content, summary, attachments, date_rank)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (detail_url, SITE_NAME, full_title, pub_date, content, source_text, attachments_str, date_rank))
            total += 1
            print("OK")
        except Exception as e:
            print(f"DB ERROR: {e}")

        time.sleep(0.3)

    conn.commit()
    conn.close()
    print(f"\n总计: {total}条")

if __name__ == "__main__":
    crawl()
