#!/usr/bin/env python3
"""
Crawler for 云浮市云安区-通知公告
CMS: NFCMS/Guangdong government CMS, UTF-8
Wraps each <p> as a separate paragraph segment for clean spacing.
"""
import requests, re, json, sys, os
from datetime import datetime, timedelta
from bs4 import BeautifulSoup, Tag
from urllib.parse import urljoin

BASE_URL = "http://www.yunan.gov.cn"
LIST_PATH = "/yaqrmzf/jcxxgk/tzgg"
LIST_URL = BASE_URL + LIST_PATH + "/"
SITE_NAME = "云浮市云安区-通知公告"
INCREMENTAL_DAYS = 7
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 20
session = requests.Session()
session.headers.update(HEADERS)


def fetch(url):
    resp = session.get(url, timeout=TIMEOUT, allow_redirects=True)
    resp.encoding = "utf-8"
    return resp.text


def parse_list(html):
    """Parse list — find all content/post_xxxxxx.html links with dates."""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    
    # Find all article links with content/post pattern
    for a in soup.find_all("a", href=True):
        href = a["href"].strip()
        if "/content/post_" not in href:
            continue
        title = a.get("title") or a.get_text(strip=True)
        if not title:
            continue
        # Resolve URL
        full_url = urljoin(BASE_URL, href)
        
        # Find date near this link
        date = ""
        parent = a.parent
        # Look up to 3 levels for a date text
        for _ in range(3):
            if parent:
                text = parent.get_text()
                m = re.search(r"(\d{4}-\d{2}-\d{2})", text)
                if m:
                    date = m.group(1)
                    break
                parent = parent.parent
        
        items.append((title.strip(), full_url, date))
    
    # Deduplicate by URL
    seen = set()
    unique = []
    for t, u, d in items:
        if u not in seen:
            seen.add(u)
            unique.append((t, u, d))
    
    return unique


def parse_detail(html, url):
    """Parse detail — div.nrcon > div.con_contitle + div.con_fbtime + div.con_article#zoomcon"""
    soup = BeautifulSoup(html, "html.parser")

    # Title
    title = ""
    title_div = soup.find("div", class_="con_contitle")
    if title_div:
        title = title_div.get_text(strip=True)

    # Date
    pub_date = ""
    fbtime = soup.find("div", class_="con_fbtime")
    if fbtime:
        text = fbtime.get_text()
        m = re.search(r"(\d{4}-\d{2}-\d{2})", text)
        if m:
            pub_date = m.group(1)

    # Content from div.con_article#zoomcon — wrap each <p> as its own segment
    content_html = ""
    content_div = soup.find("div", class_="con_article", id="zoomcon")
    if not content_div:
        content_div = soup.find("div", class_="con_article")
    if not content_div:
        content_div = soup.find("div", class_="con_leftt")

    if content_div:
        parts = []
        processed = set()
        for child in content_div.find_all(["p", "div", "table", "img"], recursive=True):
            if id(child) in processed:
                continue
            processed.add(id(child))
            tag = child.name.lower()

            if tag == "div" and not child.find(["p", "table", "img", "a"]):
                continue

            if tag == "p":
                # Each <p> is one paragraph segment — extract text preserving <a>/<img>
                p_parts = []
                for elem in child.children:
                    if isinstance(elem, str):
                        txt = elem.strip()
                        if txt:
                            p_parts.append(txt)
                    elif isinstance(elem, Tag):
                        processed.add(id(elem))
                        if elem.name == "a":
                            ahref = elem.get("href", "").strip()
                            atxt = elem.get_text(strip=True)
                            if ahref and atxt:
                                full_href = urljoin(url, ahref)
                                p_parts.append(f'<p><a href="{full_href}">{atxt}</a></p>')
                            elif atxt:
                                p_parts.append(atxt)
                        elif elem.name == "img":
                            src = elem.get("src", "")
                            if src:
                                alt = elem.get("alt", "")
                                full_src = urljoin(url, src)
                                p_parts.append(f'<p><a href="{full_src}">查看图片</a></p>' if alt else f'<p><a href="{full_src}">查看图片</a></p>')
                        else:
                            txt = elem.get_text(" ", strip=True)
                            if txt:
                                p_parts.append(txt)
                txt = " ".join(p_parts)
                if txt:
                    parts.append(txt)
            elif tag == "div":
                div_parts = []
                for a in child.find_all("a", href=True):
                    processed.add(id(a))
                    ahref = a.get("href", "").strip()
                    atxt = a.get_text(strip=True)
                    if ahref and atxt:
                        full_href = urljoin(url, ahref)
                        div_parts.append(f'<p><a href="{full_href}">{atxt}</a></p>')
                txt = child.get_text(" ", strip=True)
                if txt and div_parts:
                    parts.append(txt)
                    parts.extend(div_parts)
                elif txt:
                    parts.append(txt)
            elif tag == "table":
                table_rows = []
                for tr in child.find_all("tr"):
                    cells = [td.get_text(" ", strip=True) for td in tr.find_all(["td", "th"])]
                    if cells:
                        table_rows.append("| " + " | ".join(cells) + " |")
                if table_rows:
                    parts.append("\n".join(table_rows))
            elif tag == "img":
                src = child.get("src", "")
                if src:
                    alt = child.get("alt", "")
                    full_src = urljoin(url, src)
                    parts.append(f'<p><a href="{full_src}">查看图片</a></p>' if alt else f'<p><a href="{full_src}">查看图片</a></p>')
        content_html = "\n\n".join(parts)

    # Attachments
    scope = content_div or soup
    attachments = []
    for a in scope.find_all("a", href=True):
        ahref = a["href"].strip().lower()
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|txt)$", ahref):
            full_url = urljoin(url, a["href"].strip())
            attachments.append({"title": a.get_text(strip=True) or full_url.split("/")[-1], "url": full_url})

    if len(content_html.strip()) < 20 and attachments:
        content_html = f'<p><a href="{url}">{title}</a></p>\n\n附件列表：\n'
        for att in attachments:
            content_html += f"  - [{att['title']}]({att['url']})\n"

    return title, pub_date, content_html, attachments


def incremental_filter(items):
    cutoff = datetime.now() - timedelta(days=INCREMENTAL_DAYS)
    filtered = []
    for title, url, date_str in items:
        try:
            item_date = datetime.strptime(date_str, "%Y-%m-%d")
            if item_date >= cutoff:
                filtered.append((title, url, date_str))
        except (ValueError, IndexError):
            filtered.append((title, url, date_str))
    return filtered


def main():
    is_incremental = any(arg in sys.argv for arg in ["--incremental", "incremental", "inc"])
    page_urls = [LIST_URL]
    for i in range(2, MAX_PAGES + 1):
        page_urls.append(BASE_URL + LIST_PATH + f"/index_{i}.html")

    all_items = []
    for idx, url in enumerate(page_urls):
        try:
            html = fetch(url)
            items = parse_list(html)
            print(f"Page {idx+1}: {len(items)} items", file=sys.stderr)
            all_items.extend(items)
        except Exception as e:
            print(f"Page {idx+1} error ({url}): {e}", file=sys.stderr)

    print(f"Total: {len(all_items)}", file=sys.stderr)
    if is_incremental:
        all_items = incremental_filter(all_items)
        print(f"Incremental: {len(all_items)}", file=sys.stderr)

    seen = set()
    unique_items = []
    for item in all_items:
        if item[1] not in seen:
            seen.add(item[1])
            unique_items.append(item)

    print(f"Unique: {len(unique_items)}", file=sys.stderr)

    results = []
    for title, url, list_date in unique_items:
        try:
            html = fetch(url)
            det_title, det_date, content, attachments = parse_detail(html, url)
            final_title = det_title or title
            final_date = det_date or list_date
            results.append({
                "title": final_title,
                "page_url": url,
                "publish_date": final_date,
                "content": content,
                "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
                "site_name": SITE_NAME,
            })
            print(f"  OK: {final_title[:50]}", file=sys.stderr)
        except Exception as e:
            print(f"  ERR {url}: {e}", file=sys.stderr)

    for r in results:
        print(json.dumps(r, ensure_ascii=False))


if __name__ == "__main__":
    main()
