#!/usr/bin/env python3
"""广州市黄埔区-建设项目环境影响评价信息 爬虫"""
import re
import json
import time
import hashlib
import requests
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "https://www.hp.gov.cn"
LIST_URL = f"{BASE_URL}/hpqgzkfqzdlyzl/hjbh/jsxmhjyxpjxx/index.html"
SITE_NAME = "广州市黄埔区-建设项目环境影响评价信息"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def parse_list_page(html):
    """Parse list page, return list of (url, title, date_str)"""
    items = []
    # <li><a href="..." title="FULL TITLE" target="_blank">TEXT</a><span>YYYY-MM-DD</span></li>
    li_pattern = re.compile(r'<li[^>]*>(.*?)</li>', re.DOTALL)
    for m in li_pattern.finditer(html):
        li = m.group(1)
        if '/content/post_' not in li:
            continue
        a_m = re.search(r'<a[^>]*href="([^"]+)"[^>]*title="([^"]*)"', li)
        date_m = re.search(r'<span>(\d{4}-\d{2}-\d{2})</span>', li)
        if a_m:
            href = a_m.group(1).strip()
            title = a_m.group(2).strip()
            date_str = date_m.group(1).strip() if date_m else ""
            items.append((href, title, date_str))
    return items


def parse_detail(html, url):
    """Parse detail page"""
    soup = BeautifulSoup(html, "html.parser")

    # Title from <title> tag
    title_tag = soup.find("title")
    full_title = ""
    if title_tag:
        t = title_tag.get_text(strip=True)
        if t:
            # Remove site suffix (two titles separated by comma)
            parts = t.split("_")
            if parts:
                full_title = parts[0].strip()
        # Also try h1
        if not full_title:
            h1 = soup.find("h1")
            if h1:
                full_title = h1.get_text(strip=True)

    # Content - <div class="content" id="zhengwen"> -> <div id="zoomcon"> -> <p>
    content_div = soup.find("div", class_="content", id="zhengwen")
    if not content_div:
        content_div = soup.find("div", class_="content")
    if not content_div:
        content_div = soup.find("div", class_="print-class")
    if not content_div:
        # Fallback
        all_divs = soup.find_all("div")
        for d in all_divs:
            text = d.get_text(strip=True)
            classes = d.get("class", [])
            if len(text) > 300 and any(c in str(classes) for c in ["content", "article", "main"]):
                content_div = d
                break

    content_html = ""
    attachments = []

    if content_div:
        # Find the actual article body - zoomcon div
        zoomcon = content_div.find("div", id="zoomcon")
        if zoomcon:
            content_div = zoomcon

        # Extract attachments from all links in the page
        for a_tag in soup.find_all("a"):
            href = a_tag.get("href", "")
            a_text = a_tag.get_text(strip=True)
            if re.search(r"\.(doc|docx|pdf|xls|xlsx|zip|rar)$", href, re.I):
                if not href.startswith("http"):
                    href = f"{BASE_URL}{href}" if href.startswith("/") else f"{BASE_URL}/{href}"
                if len(a_text) > 2:
                    attachments.append({"title": a_text, "url": href})

        # Handle tables
        for table in content_div.find_all("table"):
            table_html = str(table)
            table.replace_with(f"\n\n[TABLE]\n{table_html}\n[/TABLE]\n")

        # Get paragraphs
        content_parts = []
        for child in content_div.children:
            tag_name = getattr(child, "name", None)
            if tag_name == "p":
                text = body_text(child).strip()
                if text:
                    content_parts.append(text)
            elif tag_name == "div":
                text = body_text(child).strip()
                if text:
                    content_parts.append(text)
            elif child.name is None:
                text = str(child).strip()
                if text:
                    content_parts.append(text)

        if not content_parts:
            content_html = body_text(content_div)
        else:
            content_html = "\n\n".join(content_parts)

        # Restore TABLE markers
        content_html = re.sub(
            r"\[TABLE\]\n(.*?)\n\[/TABLE\]",
            lambda m: f"\n\n{m.group(1)}\n\n",
            content_html,
            flags=re.DOTALL,
        )

        # Filter metadata lines
        metadata_patterns = [
            r"^责任编辑[：:]", r"^初审[：:]", r"^复审[：:]", r"^终审[：:]",
            r"^\[纠错\]", r"^【纠错】", r"^扫一扫在手机打开当前页",
        ]
        lines = content_html.split("\n")
        filtered_lines = []
        for line in lines:
            stripped = line.strip()
            skip = False
            for pat in metadata_patterns:
                if re.match(pat, stripped):
                    skip = True
                    break
            if not skip:
                filtered_lines.append(line)
        content_html = "\n".join(filtered_lines).strip()

        # Clean up lines with only script content
        content_html = re.sub(r'var\s+\w+\s*=\s*"[^"]*"\s*', '', content_html)
        content_html = re.sub(r'\n\s*\n\s*\n', '\n\n', content_html)

        # PDF empty content fallback
        if len(content_html.strip()) < 20:
            content_html = f'<p><a href="{url}">{full_title}</a></p>\nPDF附件见原文链接'
            if attachments:
                for att in attachments:
                    content_html += f"\n📎 {att['title']}: {att['url']}"

    # Publish date from meta
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]
    if not pub_date:
        date_matches = re.findall(r"(\d{4}-\d{1,2}-\d{1,2})", html)
        if date_matches:
            pub_date = date_matches[0]

    return {
        "title": full_title,
        "content": content_html,
        "publish_date": pub_date,
        "attachments": json.dumps(attachments, ensure_ascii=False),
        "source_url": url,
    }


def main():
    print(f"[列表] {LIST_URL}", flush=True)

    try:
        resp = session.get(LIST_URL, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            print(f"  -> HTTP {resp.status_code}", flush=True)
            return
    except Exception as e:
        print(f"  -> 请求失败: {e}", flush=True)
        return

    items = parse_list_page(resp.text)
    print(f"  -> 解析到 {len(items)} 条", flush=True)

    if not items:
        print("  -> 无内容", flush=True)
        return

    results = []
    for idx, (item_url, list_title, date_str) in enumerate(items, 1):
        print(f"  [{idx}/{len(items)}] {list_title[:60]}...", flush=True)
        try:
            dresp = session.get(item_url, timeout=30)
            dresp.encoding = "utf-8"
            if dresp.status_code != 200:
                print(f"    -> HTTP {dresp.status_code}", flush=True)
                continue
        except Exception as e:
            print(f"    -> 详情页请求失败: {e}", flush=True)
            time.sleep(1)
            continue

        detail = parse_detail(dresp.text, item_url)
        if detail["title"]:
            list_title = detail["title"]
        detail["list_title"] = list_title
        detail["list_date"] = date_str
        results.append(detail)
        time.sleep(0.5)

    # Output
    output_path = f"/root/gov_crawler/output/hp_hj_{datetime.now().strftime('%Y%m%d_%H%M%S')}.jsonl"
    count = 0
    with open(output_path, "w", encoding="utf-8") as f:
        for r in results:
            item_id = hashlib.md5(r["source_url"].encode()).hexdigest()
            record = {
                "id": item_id,
                "title": r["title"] or r["list_title"],
                "site_name": SITE_NAME,
                "source_url": r["source_url"],
                "publish_date": r["publish_date"] or r["list_date"],
                "content": r["content"],
                "attachments": r["attachments"],
                "created_at": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
                "updated_at": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
            }
            f.write(json.dumps(record, ensure_ascii=False) + "\n")
            count += 1
    print(f"\n输出 {count} 条到 {output_path}", flush=True)
    print(f"DONE: {count}条", flush=True)


if __name__ == "__main__":
    main()
