#!/usr/bin/env python3
"""
唐山行政审批局-通知公告 (xzspj.tangshan.gov.cn)
CMS: PHPCMS
分页: ?page=N (15条/页)
爬取: 前5页 → 约75条 (增量模式只爬第1页)
"""

import requests
import re
import json
import sys
import os
from datetime import datetime, date
from bs4 import BeautifulSoup

INCREMENTAL = "--incremental" in sys.argv
BASE_URL = "https://xzspj.tangshan.gov.cn"
LIST_URL = BASE_URL + "/index.php?m=content&c=index&a=lists&catid=29"
MAX_PAGES = 1 if INCREMENTAL else 5
SITE_NAME = "唐山市行政审批局-通知公告"
GROUP = "政务"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": LIST_URL,
}

visited = {}  # url -> item

headers = HEADERS.copy()
headers["Referer"] = LIST_URL

for page in range(1, MAX_PAGES + 1):
    if page == 1:
        url = LIST_URL
    else:
        url = LIST_URL + f"&page={page}"

    print(f"\n--- Page {page}: {url}")
    try:
        r = requests.get(url, headers=headers, verify=False, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  FAIL: {e}")
        continue

    # Parse list items
    soup = BeautifulSoup(r.text, "html.parser")

    # Find items: li > a with show links
    items_found = 0
    for li in soup.select("li"):
        a = li.find("a", href=re.compile(r"show&catid=29"))
        if not a:
            continue

        href = a.get("href", "").strip()
        if not href.startswith("http"):
            href = BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href

        # Extract date from the text after the link
        li_text = li.get_text(" ", strip=True)
        date_match = re.search(r"(\d{4}-\d{2}-\d{2})", li_text)

        display_title = a.get_text(strip=True)
        if not display_title or len(display_title) < 5:
            continue

        items_found += 1
        if href not in visited:
            visited[href] = {
                "url": href,
                "display_title": display_title,
                "page_date": date_match.group(1) if date_match else "",
            }
            print(f"  [{items_found}] {display_title[:50]}... | {visited[href]['page_date']}")

    if items_found == 0:
        print(f"  -> 0 items found, stopping")
        break

print(f"\nTotal unique items from list: {len(visited)}")

# Now visit each detail page for full title and content
results = []
session = requests.Session()
session.headers.update(headers)

for idx, (url, item) in enumerate(visited.items(), 1):
    print(f"\n[{idx}/{len(visited)}] Fetching detail: {item['display_title'][:40]}...")
    try:
        r = session.get(url, verify=False, timeout=30)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  FAIL: {e}")
        continue

    detail_soup = BeautifulSoup(html, "html.parser")

    # Full title from h3 in main_c
    full_title = ""
    h3 = detail_soup.select_one("div.main_c h3")
    if h3:
        full_title = h3.get_text(" ", strip=True)

    # Fallback: title from the list if h3 fails
    if not full_title:
        full_title = item["display_title"]

    # Date from ondate div
    pub_date = ""
    ondate = detail_soup.select_one("div.ondate")
    if ondate:
        date_match = re.search(r"(\d{4}-\d{2}-\d{2})", ondate.get_text())
        if date_match:
            pub_date = date_match.group(1)
    if not pub_date:
        pub_date = item["page_date"]

    # Content from main_c (exclude h3 and ondate)
    content_html = ""
    main_c = detail_soup.select_one("div.main_c")
    if main_c:
        # Clone to avoid mutating the soup
        content_parts = []
        for child in main_c.children:
            if hasattr(child, "name"):
                if child.name in ("h3",):
                    continue
                if child.name == "div" and "ondate" in child.get("class", []):
                    continue
                content_parts.append(str(child))
            elif child.strip():
                content_parts.append(str(child))
        content_html = "\n".join(content_parts)

    # Clean content to text with preserved paragraph breaks
    content_text = ""
    if content_html:
        import html as html_mod
        text = content_html
        # Replace block-level tags with newlines
        text = re.sub(r'</?(?:p|div|h[1-6]|li|tr|blockquote|section|article|table|br\s*/?)[^>]*>', '\n', text, flags=re.IGNORECASE)
        # Remove all remaining HTML tags
        text = re.sub(r'<[^>]+>', '', text)
        # Decode HTML entities
        text = html_mod.unescape(text)
        # Collapse horizontal whitespace (tabs, multiple spaces)
        text = re.sub(r'[ \t]+', ' ', text)
        # Normalize newlines: max 2 consecutive
        text = re.sub(r'\n{3,}', '\n\n', text)
        content_text = text.strip()

    if not content_text:
        content_text = item["display_title"]

    # Attachments
    attachments = []
    for a_tag in detail_soup.select("a[href$='.pdf'], a[href$='.doc'], a[href$='.docx'], a[href$='.xls'], a[href$='.xlsx'], a[href$='.zip'], a[href$='.rar']"):
        attach_url = a_tag.get("href", "").strip()
        if attach_url and not attach_url.startswith("http"):
            attach_url = BASE_URL + attach_url if attach_url.startswith("/") else BASE_URL + "/" + attach_url
        attach_title = a_tag.get_text(" ", strip=True) or os.path.basename(attach_url)
        if attach_url and attach_url not in [a.get("url") for a in attachments]:
            attachments.append({"title": attach_title, "url": attach_url})

    # Handle PDF-only content (if content is very short, embed the title link)
    clean_content = re.sub(r'\s+', ' ', content_text).strip()
    if len(clean_content) < 20 and attachments:
        content_text = f'<p><a href="{url}">{full_title}</a></p>\n\n附件：\n' + "\n".join(f'<p><a href="{a["url"]}">{a["title"]}</a></p>' for a in attachments)

    results.append({
        "title": full_title,
        "page_url": url,
        "content": content_text,
        "publish_date": pub_date,
        "site_name": SITE_NAME,
        "group": GROUP,
        "attachments": attachments if attachments else [],
    })

    print(f"  ✓ {full_title[:50]} | {pub_date} | {'+attachments' if attachments else ''}")

# Deduplicate by URL
seen_urls = set()
unique_results = []
for r in results:
    if r["page_url"] not in seen_urls:
        seen_urls.add(r["page_url"])
        unique_results.append(r)

print(f"\n=== Summary ===")
print(f"Total items: {len(unique_results)}")

# Save to JSONL
output_file = f"/root/gov_crawler/output/xzspj_tangshan_{datetime.now().strftime('%Y%m%d_%H%M%S')}.jsonl"
os.makedirs("/root/gov_crawler/output", exist_ok=True)

with open(output_file, "w", encoding="utf-8") as f:
    for r in unique_results:
        f.write(json.dumps(r, ensure_ascii=False) + "\n")

print(f"Saved to: {output_file}")

# Import to DB
import subprocess
import_path = "/root/gov_crawler/import_jsonl.py"
if os.path.exists(import_path):
    cmd = f"cd /root/gov_crawler && python3 {import_path} {output_file}"
    result = subprocess.run(cmd, shell=True, capture_output=True, text=True)
    print(result.stdout)
    if result.stderr:
        print("STDERR:", result.stderr[:500])
else:
    print("WARN: import script not found at", import_path)

print("Done!")
