#!/usr/bin/env python3
"""
Crawl 广东省国际工程咨询有限公司 - 公示公告
https://www.gdiecc.com.cn/gongshigonggao-list.html
WAF protected by Aliyun - uses Playwright for cookie bootstrapping, then requests for all pages.

List: <ul><li><a href="/gongshigonggao-show-{id}.html">title</a><span>更新时间：YYYY-MM-DD</span></li></ul>
Detail: 
  Title: <h1> tag text
  Date: 发布时间：YYYY-MM-DD in .wh-article-top
  Content: .wh-article-con div
  Attachments: <a href="*.pdf|*.doc|*.docx">
Pagination: /gongshigonggao-list-{page}.html (page 1 = no suffix)
"""
import asyncio, requests, json, re, os, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from playwright.async_api import async_playwright

BASE_URL = "https://www.gdiecc.com.cn"
LIST_URL = "https://www.gdiecc.com.cn/gongshigonggao-list.html"
OUTPUT_FILE = "/root/gov_crawler/output/gdiecc.jsonl"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

async def bootstrap_cookies():
    """Use Playwright once to bypass WAF and get valid cookies."""
    async with async_playwright() as p:
        browser = await p.chromium.launch(
            headless=True,
            executable_path="/root/.cache/ms-playwright/chromium-1228/chrome-linux64/chrome",
            args=["--no-sandbox", "--disable-setuid-sandbox"]
        )
        ctx = await browser.new_context(
            user_agent=HEADERS["User-Agent"],
            locale="zh-CN"
        )
        pg = await ctx.new_page()
        await pg.goto(LIST_URL, wait_until="networkidle", timeout=60000)
        await asyncio.sleep(3)
        cookies = {c["name"]: c["value"] for c in await ctx.cookies()}
        await browser.close()
        return cookies

def get_page_url(page_num):
    if page_num == 1:
        return LIST_URL
    return f"https://www.gdiecc.com.cn/gongshigonggao-list-{page_num}.html"

def parse_list(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for ul in soup.find_all("ul"):
        lis = ul.find_all("li")
        if len(lis) < 5:
            continue
        for li in lis:
            a = li.find("a", href=True)
            if not a:
                continue
            href = a.get("href", "")
            if "gongshigonggao-show" not in href:
                continue
            if not href.startswith("http"):
                href = urljoin(BASE_URL, href)
            title = a.get_text(strip=True)
            if not title:
                continue
            date = ""
            span = li.find("span")
            if span:
                txt = span.get_text(strip=True)
                m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
                if m:
                    date = m.group(1)
            items.append({"url": href, "title": title, "date": date})
    return items

def parse_detail(url, html):
    soup = BeautifulSoup(html, "html.parser")
    
    # Title from h1
    title = ""
    top_div = soup.find("div", class_="wh-article-top")
    if top_div:
        h1 = top_div.find("h1")
    else:
        h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    
    # Date from .wh-article-top
    pub_date = ""
    date_div = soup.find("div", class_="wh-article-top")
    if date_div:
        txt = date_div.get_text(strip=True)
        m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
        if m:
            pub_date = m.group(1)
    
    # Also check .ingongshi
    if not pub_date:
        for cls in ["ingongshi", "wh-article"]:
            el = soup.find("div", class_=cls)
            if el:
                txt = el.get_text(strip=True)
                m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", txt)
                if m:
                    pub_date = m.group(1)
                    break
    
    # Content from .wh-article-con
    content_div = soup.find("div", class_="wh-article-con")
    if not content_div:
        content_div = soup.find("div", class_="content")
    
    content_html = ""
    summary = ""
    attachments = []
    if content_div:
        # Keep table structure, clean styles
        for tag in content_div.find_all(True):
            attrs_to_keep = ["href", "src", "alt", "target", "border"]
            for attr in list(tag.attrs):
                if attr not in attrs_to_keep and not attr.startswith("mso-"):
                    del tag[attr]
        content_html = str(content_div)
        summary = content_div.get_text("\n", strip=True)
        # Attachments
        for a in content_div.find_all("a", href=True):
            ahref = a.get("href", "")
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", ahref.lower()):
                full_url = ahref if ahref.startswith("http") else urljoin(url, ahref)
                attachments.append({"url": full_url, "text": a.get_text(strip=True) or os.path.basename(ahref)})
    
    return {
        "title": title,
        "pub_date": pub_date,
        "content_html": content_html,
        "summary": summary[:500] if summary else "",
        "attachments": attachments,
    }

def main():
    # Bootstrap WAF cookies via Playwright
    print("Bootstrapping WAF cookies via Playwright...")
    cookies = asyncio.run(bootstrap_cookies())
    print("Got cookies:", {k: v[:15]+"..." for k, v in cookies.items() if len(v) > 15})
    
    s = requests.Session()
    s.headers.update(HEADERS)
    s.cookies.update(cookies)
    
    # Page 1
    print("Fetching page 1...")
    resp = s.get(LIST_URL, timeout=30)
    resp.encoding = "utf-8"
    items = parse_list(resp.text)
    print(f"  Found {len(items)} items on page 1")
    
    # Find total pages
    total_pages = 1
    for p in range(2, 101):
        url = get_page_url(p)
        try:
            r = s.get(url, timeout=30)
            r.encoding = "utf-8"
            if "renderData" in r.text:
                print(f"  Page {p}: WAF blocked")
                break
            page_items = parse_list(r.text)
            if len(page_items) == 0:
                print(f"  Page {p}: 0 items, end")
                break
            items.extend(page_items)
            total_pages = p
            if p % 20 == 0:
                print(f"  Page {p}: {len(page_items)} items (total: {len(items)})")
            time.sleep(0.3)
        except Exception as e:
            print(f"  Page {p}: ERROR {e}")
            break
    
    print(f"\nTotal items from list: {len(items)}, pages: {total_pages}")
    
    # Fetch detail pages
    os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)
    record_count = 0
    with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
        for i, item in enumerate(items):
            preview = item["title"][:40] + "..." if len(item["title"]) > 40 else item["title"]
            print(f"  [{i+1}/{len(items)}] {preview}")
            try:
                resp = s.get(item["url"], timeout=30)
                resp.encoding = "utf-8"
                if "renderData" in resp.text:
                    print(f"    WAF blocked on detail! Re-bootstrapping...")
                    cookies = asyncio.run(bootstrap_cookies())
                    s.cookies.update(cookies)
                    resp = s.get(item["url"], timeout=30)
                    resp.encoding = "utf-8"
                
                detail = parse_detail(item["url"], resp.text)
                final_title = detail["title"] or item["title"]
                record = {
                    "title": final_title,
                    "url": item["url"],
                    "date": detail["pub_date"] or item["date"],
                    "content": detail["content_html"],
                    "summary": detail["summary"],
                    "site_name": "广东省国际工程咨询有限公司",
                    "group": "企业",
                    "attachments": json.dumps(detail["attachments"], ensure_ascii=False) if detail["attachments"] else "",
                }
                f.write(json.dumps(record, ensure_ascii=False) + "\n")
                record_count += 1
            except Exception as e:
                print(f"    [ERROR] {item['url']}: {e}")
            time.sleep(0.5)
    
    print(f"\nDone! {record_count} records written to {OUTPUT_FILE}")

if __name__ == "__main__":
    if "--incremental" in sys.argv:
        print("Incremental mode: crawl page 1 only")
        cookies = asyncio.run(bootstrap_cookies())
        s = requests.Session()
        s.headers.update(HEADERS)
        s.cookies.update(cookies)
        resp = s.get(LIST_URL, timeout=30)
        resp.encoding = "utf-8"
        items = parse_list(resp.text)
        print(f"Found {len(items)} items on page 1")
        os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)
        with open(OUTPUT_FILE, "w", encoding="utf-8") as f:
            for item in items:
                try:
                    resp = s.get(item["url"], timeout=30)
                    resp.encoding = "utf-8"
                    detail = parse_detail(item["url"], resp.text)
                    final_title = detail["title"] or item["title"]
                    record = {
                        "title": final_title,
                        "url": item["url"],
                        "date": detail["pub_date"] or item["date"],
                        "content": detail["content_html"],
                        "summary": detail["summary"],
                        "site_name": "广东省国际工程咨询有限公司",
                        "group": "企业",
                        "attachments": json.dumps(detail["attachments"], ensure_ascii=False) if detail["attachments"] else "",
                    }
                    f.write(json.dumps(record, ensure_ascii=False) + "\n")
                except Exception as e:
                    print(f"ERROR: {e}")
                time.sleep(0.5)
        print(f"Incremental done: {len(items)} records")
    else:
        main()
