#!/usr/bin/env python3
"""
萧县人民政府 - 两个栏目爬虫
1. 圣泉镇-重大行政决策预公开 (column 263, catId=24260769)
2. 萧县经开区-意见征集 (column 239, catId=3846066)
CMS: 政府网站 (萧县人民政府)
分页: ?pageIndex=N (15条/页, 爬前5页)
"""
import requests
import re
import json
import sys
import os
from datetime import datetime
from bs4 import BeautifulSoup

INCREMENTAL = "--incremental" in sys.argv
MAX_PAGES = 1 if INCREMENTAL else 5

SITES = [
    {
        "name": "萧县圣泉镇-重大行政决策预公开",
        "list_url": "https://www.ahxx.gov.cn/public/column/263?type=4&catId=24260769&action=list",
        "group": "政务",
        "column": "263",
        "cat_id": "24260769",
    },
    {
        "name": "萧县经开区-意见征集",
        "list_url": "https://www.ahxx.gov.cn/public/column/239?type=4&catId=3846066&action=list",
        "group": "政务",
        "column": "239",
        "cat_id": "3846066",
    },
]

BASE_URL = "https://www.ahxx.gov.cn"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(HEADERS)

all_results = []

for site in SITES:
    print(f"\n{'='*60}")
    print(f"=== {site['name']} ===")
    print(f"{'='*60}")

    # Fetch list pages (MAX_PAGES pages)
    items_on_page = []
    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            url = site["list_url"]
        else:
            url = site["list_url"] + f"&pageIndex={page}"

        print(f"  Page {page}: {url}")
        try:
            r = session.get(url, verify=False, timeout=30)
            r.encoding = "utf-8"
            html = r.text
        except Exception as e:
            print(f"    FAIL: {e}")
            continue

        # Extract article links with title attribute
        found = 0
        for m in re.finditer(r'href="(/public/{}/(\d+)\.html)"'.format(site["column"]), html):
            pos = m.start()
            href = m.group(1)
            article_id = m.group(2)
            # Context around the href to find title and date
            ctx = html[max(0, pos - 200):pos + 300]
            title_m = re.search(r'title="([^"]*)"', ctx)
            title_from_list = title_m.group(1).strip() if title_m else ""
            # Find date in before context
            dates = re.findall(r'(\d{4}-\d{2}-\d{2})', html[max(0, pos - 300):pos])
            date_str = dates[-1] if dates else ""
            # Skip items without meaningful title
            if not title_from_list or len(title_from_list) < 5:
                continue
            # Deduplicate by article_id
            if not any(item["id"] == article_id for item in items_on_page):
                items_on_page.append({
                    "id": article_id,
                    "url": BASE_URL + href,
                    "title": title_from_list,
                    "date": date_str,
                })
                found += 1

        print(f"    Found {found} new items on page {page}")

    print(f"  Total items: {len(items_on_page)}")
    for i, item in enumerate(items_on_page):
        print(f"    [{i+1}] {item['title'][:50]}... | {item['date']}")

    # Fetch each detail page
    for idx, item in enumerate(items_on_page, 1):
        print(f"\n  [{idx}/{len(items_on_page)}] Fetching: {item['title'][:40]}...")
        try:
            r = session.get(item["url"], verify=False, timeout=30)
            r.encoding = "utf-8"
            detail_html = r.text
        except Exception as e:
            print(f"    FAIL: {e}")
            continue

        detail_soup = BeautifulSoup(detail_html, "html.parser")

        # Full title from h1
        full_title = ""
        h1 = detail_soup.select_one("h1")
        if h1:
            full_title = h1.get_text(" ", strip=True)
        if not full_title:
            full_title = item["title"]

        # Date from detail page metadata
        pub_date = item["date"]
        secnr = detail_soup.select_one("div.secnr")
        if secnr:
            dm = re.search(r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})", secnr.get_text())
            if dm:
                pub_date = dm.group(1)

        # Content from .wzcon.clearfix
        content_text = ""
        wzcon = detail_soup.select_one("div.wzcon.clearfix")
        if not wzcon:
            wzcon = detail_soup.select_one("div.wzcon")

        if wzcon:
            # Convert HTML to clean text with paragraph breaks
            html_str = str(wzcon)
            # Replace block-level tags with newlines
            text = re.sub(
                r'</?(?:p|div|h[1-6]|li|tr|blockquote|section|article|table|br\s*/?)[^>]*>',
                "\n", html_str, flags=re.IGNORECASE
            )
            # Remove remaining HTML tags
            text = re.sub(r'<[^>]+>', '', text)
            # Decode entities
            import html as html_mod
            text = html_mod.unescape(text)
            # Collapse horizontal whitespace
            text = re.sub(r'[ \t]+', ' ', text)
            # Normalize newlines
            text = re.sub(r'\n{3,}', '\n\n', text)
            content_text = text.strip()

        if not content_text:
            content_text = full_title

        # Attachments
        attachments = []
        for a_tag in detail_soup.select("a[href$='.pdf'], a[href$='.doc'], a[href$='.docx'], a[href$='.xls'], a[href$='.xlsx'], a[href$='.zip'], a[href$='.rar']"):
            attach_url = a_tag.get("href", "").strip()
            if attach_url and not attach_url.startswith("http"):
                attach_url = BASE_URL + attach_url if attach_url.startswith("/") else BASE_URL + "/" + attach_url
            attach_title = a_tag.get_text(" ", strip=True) or os.path.basename(attach_url)
            if attach_url and attach_url not in [a.get("url") for a in attachments]:
                attachments.append({"title": attach_title, "url": attach_url})

        # Handle empty content (PDF-only)
        clean = re.sub(r'\s+', ' ', content_text).strip()
        if len(clean) < 20 and attachments:
            content_text = f'<p><a href="{item["url"]}">{full_title}</a></p>\n\n附件：\n' + "\n".join(f'<p><a href="{a["url"]}">{a["title"]}</a></p>' for a in attachments)

        result = {
            "title": full_title,
            "page_url": item["url"],
            "content": content_text,
            "publish_date": pub_date,
            "site_name": site["name"],
            "group": site["group"],
            "attachments": attachments if attachments else [],
        }
        all_results.append(result)
        print(f"    ✓ {full_title[:50]} | {pub_date} | {'+attachments' if attachments else ''}")

print(f"\n{'='*60}")
print(f"Total items: {len(all_results)}")

# Save to JSONL
output_file = f"/root/gov_crawler/output/ahxx_{datetime.now().strftime('%Y%m%d_%H%M%S')}.jsonl"
os.makedirs("/root/gov_crawler/output", exist_ok=True)

with open(output_file, "w", encoding="utf-8") as f:
    for r in all_results:
        f.write(json.dumps(r, ensure_ascii=False) + "\n")

print(f"Saved: {output_file}")

# Import to DB
import subprocess
import_path = "/root/gov_crawler/import_jsonl.py"
if os.path.exists(import_path):
    cmd = f"cd /root/gov_crawler && python3 {import_path} {output_file}"
    result = subprocess.run(cmd, shell=True, capture_output=True, text=True)
    print(result.stdout)
    if result.stderr:
        print("STDERR:", result.stderr[:500])
else:
    print("WARN: import script not found")

print("Done!")
