#!/usr/bin/env python3
"""
Crawl 广东省国际工程咨询有限公司 - 公示公告
https://www.gdiecc.com.cn/gongshigonggao-list.html
WAF protected by Aliyun - uses Playwright for cookie bootstrapping, then requests for all pages.

List: <ul><li><a href="/gongshigonggao-show-{id}.html">title</a><span>更新时间：YYYY-MM-DD</span></li></ul>
Detail: 
  Title: <h1> tag text
  Date: 发布时间：YYYY-MM-DD in .wh-article-top
  Content: .wh-article-con div
  Attachments: <a href="*.pdf|*.doc|*.docx">
Pagination: /gongshigonggao-list-{page}.html (page 1 = no suffix)
"""
import asyncio, requests, json, re, os, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from playwright.async_api import async_playwright

BASE_URL = "https://www.gdiecc.com.cn"
LIST_URL = "https://www.gdiecc.com.cn/gongshigonggao-list.html"
OUTPUT_FILE = "/root/gov_crawler/output/gdiecc.jsonl"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

async def bootstrap_cookies():
    """Use Playwright once to bypass WAF and get valid cookies."""
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path="/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome",
            args=["--no-sandbox", "--disable-setuid-sandbox"]
        )
        ctx = await browser.new_context(
            user_agent=HEADERS["User-Agent"],
            locale="zh-CN"
        )
        pg = await ctx.new_page()
        await pg.goto(LIST_URL, wait_until="networkidle", timeout=60000)
        await asyncio.sleep(3)
        cookies = {c["name"]: c["value"] for c in await ctx.cookies()}
        await browser.close()
        return cookies

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_page_url(page_num):
    if page_num == 1:
        return LIST_URL
    return f"https://www.gdiecc.com.cn/gongshigonggao-list-{page_num}.html"

def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for ul in soup.find_all("ul"):
        lis = ul.find_all("li")
        if len(lis) < 5:
            continue
        for li in lis:
            a = li.find("a", href=True)
            if not a:
                continue
            href = a.get("href", "")
            if "gongshigonggao-show" not in href:
                continue
            if not href.startswith("http"):
                href = urljoin(BASE_URL, href)
            title = a.get_text(strip=True)
            if not title:
                continue
            date = ""
            span = li.find("span")
            if span:
                txt = span.get_text(strip=True)
                m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
                if m:
                    date = m.group(1)
            items.append({"url": href, "title": title, "date": date})
    return items

def parse_detail(url, html):
    soup = BeautifulSoup(html, "html.parser")
    
    # Title from h1
    title = ""
    top_div = soup.find("div", class_="wh-article-top")
    if top_div:
        h1 = top_div.find("h1")
    else:
        h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    
    # Date from .wh-article-top
    pub_date = ""
    date_div = soup.find("div", class_="wh-article-top")
    if date_div:
        txt = date_div.get_text(strip=True)
        m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
        if m:
            pub_date = m.group(1)
    
    # Also check .ingongshi
    if not pub_date:
        for cls in ["ingongshi", "wh-article"]:
            el = soup.find("div", class_=cls)
            if el:
                txt = el.get_text(strip=True)
                m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
                if m:
                    pub_date = m.group(1)
                    break
    
    # Content from .wh-article-con
    content_div = soup.find("div", class_="wh-article-con")
    if not content_div:
        content_div = soup.find("div", class_="content")
    
    content_html = ""
    summary = ""
    attachments = []
    if content_div:
        # Keep table structure, clean styles
        for tag in content_div.find_all(True):
            attrs_to_keep = ["href", "src", "alt", "target", "border"]
            for attr in list(tag.attrs):
                if attr not in attrs_to_keep and not attr.startswith("mso-"):
                    del tag[attr]
        content_html = str(content_div)
        summary = body_text(content_div)
        # Attachments
        for a in content_div.find_all("a", href=True):
            ahref = a.get("href", "")
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", ahref.lower()):
                full_url = ahref if ahref.startswith("http") else urljoin(url, ahref)
                attachments.append({"url": full_url, "text": a.get_text(strip=True) or os.path.basename(ahref)})
    
    return {
        "title": title,
        "pub_date": pub_date,
        "content_html": content_html,
        "summary": summary[:500] if summary else "",
        "attachments": attachments,
    }

def main():
    # Bootstrap WAF cookies via Playwright
    print("Bootstrapping WAF cookies via Playwright...")
    cookies = asyncio.run(bootstrap_cookies())
    print("Got cookies:", {k: v[:15]+"..." for k, v in cookies.items() if len(v) > 15})
    
    s = requests.Session()
    s.headers.update(HEADERS)
    s.cookies.update(cookies)
    
    # Page 1
    print("Fetching page 1...")
    resp = s.get(LIST_URL, timeout=30)
    resp.encoding = "utf-8"
    items = parse_list(resp.text)
    print(f"  Found {len(items)} items on page 1")
    
    # Find total pages
    total_pages = 1
    for p in range(2, 101):
        url = get_page_url(p)
        try:
            r = s.get(url, timeout=30)
            r.encoding = "utf-8"
            if "renderData" in r.text:
                print(f"  Page {p}: WAF blocked")
                break
            page_items = parse_list(r.text)
            if len(page_items) == 0:
                print(f"  Page {p}: 0 items, end")
                break
            items.extend(page_items)
            total_pages = p
            if p % 20 == 0:
                print(f"  Page {p}: {len(page_items)} items (total: {len(items)})")
            time.sleep(0.3)
        except Exception as e:
            print(f"  Page {p}: ERROR {e}")
            break
    
    print(f"\nTotal items from list: {len(items)}, pages: {total_pages}")
    
    # Fetch detail pages
    os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)
    record_count = 0
    with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
        for i, item in enumerate(items):
            preview = item["title"][:40] + "..." if len(item["title"]) > 40 else item["title"]
            print(f"  [{i+1}/{len(items)}] {preview}")
            try:
                resp = s.get(item["url"], timeout=30)
                resp.encoding = "utf-8"
                if "renderData" in resp.text:
                    print(f"    WAF blocked on detail! Re-bootstrapping...")
                    cookies = asyncio.run(bootstrap_cookies())
                    s.cookies.update(cookies)
                    resp = s.get(item["url"], timeout=30)
                    resp.encoding = "utf-8"
                
                detail = parse_detail(item["url"], resp.text)
                final_title = detail["title"] or item["title"]
                record = {
                    "title": final_title,
                    "url": item["url"],
                    "date": detail["pub_date"] or item["date"],
                    "content": detail["content_html"],
                    "summary": detail["summary"],
                    "site_name": "广东省国际工程咨询有限公司",
                    "group": "企业",
                    "attachments": json.dumps(detail["attachments"], ensure_ascii=False) if detail["attachments"] else "",
                }
                f.write(json.dumps(record, ensure_ascii=False) + "\n")
                record_count += 1
            except Exception as e:
                print(f"    [ERROR] {item['url']}: {e}")
            time.sleep(0.5)
    
    print(f"\nDone! {record_count} records written to {OUTPUT_FILE}")

if __name__ == "__main__":
    if "--incremental" in sys.argv:
        print("Incremental mode: crawl page 1 only")
        cookies = asyncio.run(bootstrap_cookies())
        s = requests.Session()
        s.headers.update(HEADERS)
        s.cookies.update(cookies)
        resp = s.get(LIST_URL, timeout=30)
        resp.encoding = "utf-8"
        items = parse_list(resp.text)
        print(f"Found {len(items)} items on page 1")
        os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)
        with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
            for item in items:
                try:
                    resp = s.get(item["url"], timeout=30)
                    resp.encoding = "utf-8"
                    detail = parse_detail(item["url"], resp.text)
                    final_title = detail["title"] or item["title"]
                    record = {
                        "title": final_title,
                        "url": item["url"],
                        "date": detail["pub_date"] or item["date"],
                        "content": detail["content_html"],
                        "summary": detail["summary"],
                        "site_name": "广东省国际工程咨询有限公司",
                        "group": "企业",
                        "attachments": json.dumps(detail["attachments"], ensure_ascii=False) if detail["attachments"] else "",
                    }
                    f.write(json.dumps(record, ensure_ascii=False) + "\n")
                except Exception as e:
                    print(f"ERROR: {e}")
                time.sleep(0.5)
        print(f"Incremental done: {len(items)} records")
    else:
        main()
