#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Crawler for 和布克赛尔蒙古自治县人民政府 - 通知公告
CMS: 自定义，ul.list.pt20列表 + div.content_p#NewsContent正文
Site: www.xjhbk.gov.cn/xjhbk/tzgg/
"""
import requests, re, json, sys, os, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.xjhbk.gov.cn"
LIST_PATH = "/xjhbk/tzgg"
LIST_URL = BASE_URL + LIST_PATH + "/list.shtml"
SITE_NAME = "和布克赛尔县-通知公告"
GROUP = "新疆"
INCREMENTAL_DAYS = 7
MAX_PAGES = 6
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 20
session = requests.Session()
session.headers.update(HEADERS)


def fetch(url):
    resp = session.get(url, timeout=TIMEOUT)
    resp.encoding = "utf-8"
    return resp.text


def parse_list(html):
    """
    Parse list page — ul.list.pt20 > li > a[title] + span(date)
    Pagination: list.shtml ~ list_6.shtml (6 pages, 20 items/page, ~116 total)
    """
    soup = BeautifulSoup(html, "html.parser")
    items = []
    ul = soup.find("ul", class_="list")
    if not ul:
        return items
    for li in ul.find_all("li", recursive=False):
        a = li.find("a", href=True)
        if not a:
            continue
        title = a.get("title") or a.get_text(strip=True)
        href = a["href"].strip()
        if not title or not href:
            continue
        # Date from <span>
        date_span = li.find("span")
        date = date_span.get_text(strip=True) if date_span else ""
        # Normalize URL
        if href.startswith("/"):
            href = BASE_URL + href
        elif not href.startswith("http"):
            href = BASE_URL + LIST_PATH + "/" + href.lstrip("./")
        items.append((title.strip(), href, date[:10]))
    return items


def parse_detail(html, url):
    """
    Parse detail page — div.content_p#NewsContent
    Title from div.contentbox > h3
    Source from span.from
    Date from span.date
    Attachments from <a href="...pdf|doc|docx|xls|xlsx">
    """
    soup = BeautifulSoup(html, "html.parser")

    # Title from <h3> in .contentbox
    title = ""
    contentbox = soup.find("div", class_="contentbox")
    if contentbox:
        h3 = contentbox.find("h3")
        if h3:
            title = h3.get_text(strip=True)
    if not title:
        meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
        if meta_title and meta_title.get("content"):
            title = meta_title["content"].strip()

    # Source from span.from
    source = ""
    from_span = soup.find("span", class_="from")
    if from_span:
        source = from_span.get_text(strip=True)

    # Date from span.date
    pub_date = ""
    date_span = soup.find("span", class_="date")
    if date_span:
        date_text = date_span.get_text(strip=True)
        m = re.search(r"(\d{4}-\d{2}-\d{2})", date_text)
        if m:
            pub_date = m.group(1)
    # Fallback to meta PubDate
    if not pub_date:
        meta_date = soup.find("meta", attrs={"name": "PubDate"})
        if meta_date and meta_date.get("content"):
            m = re.search(r"(\d{4}-\d{2}-\d{2})", meta_date["content"])
            if m:
                pub_date = m.group(1)

    # Content — div.content_p#NewsContent with <p> preserving <a> links
    content_html = ""
    content_div = soup.find("div", class_="content_p")
    if content_div:
        parts = []
        # If the first child is a wrapper div, use its children instead
        inner = content_div
        first_child = next((c for c in content_div.children if getattr(c, "name", None)), None)
        if first_child and first_child.name == "div":
            inner = first_child
        for child in inner.children:
            if not getattr(child, "name", None):
                continue
            tag = child.name.lower()
            if tag == "p":
                p_parts = []
                for elem in child.contents:
                    if isinstance(elem, str):
                        txt = elem.strip()
                        if txt:
                            p_parts.append(txt)
                    elif elem.name == "a":
                        ahref = elem.get("href", "").strip()
                        atxt = elem.get_text(strip=True)
                        if ahref and atxt:
                            full_href = urljoin(url, ahref)
                            p_parts.append(f"[{atxt}]({full_href})")
                        elif atxt:
                            p_parts.append(atxt)
                    elif elem.name == "img":
                        src = elem.get("src", "")
                        if src:
                            alt = elem.get("alt", "")
                            full_src = urljoin(url, src)
                            p_parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
                    elif hasattr(elem, "get_text"):
                        txt = elem.get_text(" ", strip=True)
                        if txt:
                            p_parts.append(txt)
                txt = " ".join(p_parts)
                if txt:
                    parts.append(txt)
            elif tag == "table":
                # Render table as markdown
                table_rows = []
                for tr in child.find_all("tr"):
                    cells = [td.get_text(" ", strip=True) for td in tr.find_all(["td", "th"])]
                    if cells:
                        table_rows.append("| " + " | ".join(cells) + " |")
                if table_rows:
                    parts.append("\n".join(table_rows))
            elif tag == "img":
                src = child.get("src", "")
                if src:
                    alt = child.get("alt", "")
                    full_src = urljoin(url, src)
                    parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
        content_html = "\n\n".join(parts)

    # Attachments — scan content div only
    scope = content_div or soup
    attachments = []
    for a in scope.find_all("a", href=True):
        ahref = a["href"].strip().lower()
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar|txt)$", ahref):
            full_url = urljoin(url, a["href"].strip())
            attachments.append({"title": a.get_text(strip=True) or full_url.split("/")[-1], "url": full_url})

    # If content is very short but has attachments, generate summary
    if len(content_html.strip()) < 20 and attachments:
        content_html = f"[{title}]({url})\n\n附件列表：\n"
        for att in attachments:
            content_html += f"  - [{att['title']}]({att['url']})\n"

    return title, pub_date, source, content_html, attachments


def incremental_filter(items):
    cutoff = datetime.now() - timedelta(days=INCREMENTAL_DAYS)
    filtered = []
    for title, url, date_str in items:
        try:
            item_date = datetime.strptime(date_str, "%Y-%m-%d")
            if item_date >= cutoff:
                filtered.append((title, url, date_str))
        except (ValueError, IndexError):
            filtered.append((title, url, date_str))
    return filtered


def main():
    is_incremental = any(arg in sys.argv for arg in ["--incremental", "incremental", "inc"])

    # Build page URLs: list.shtml + list_2.shtml ~ list_6.shtml
    page_urls = [LIST_URL]
    for i in range(2, MAX_PAGES + 1):
        page_urls.append(BASE_URL + LIST_PATH + f"/list_{i}.shtml")

    # Fetch all list pages
    all_items = []
    for idx, url in enumerate(page_urls):
        try:
            html = fetch(url)
            items = parse_list(html)
            print(f"Page {idx+1}: {len(items)} items", file=sys.stderr)
            all_items.extend(items)
        except Exception as e:
            print(f"Page {idx+1} error ({url}): {e}", file=sys.stderr)

    print(f"Total: {len(all_items)}", file=sys.stderr)

    if is_incremental:
        all_items = incremental_filter(all_items)
        print(f"Incremental: {len(all_items)}", file=sys.stderr)

    # Deduplicate by URL
    seen = set()
    unique_items = []
    for item in all_items:
        if item[1] not in seen:
            seen.add(item[1])
            unique_items.append(item)
    print(f"Unique: {len(unique_items)}", file=sys.stderr)

    # Connect to DB
    conn = sqlite3.connect(DB_PATH)
    cur = conn.cursor()
    saved = 0
    skipped = 0
    errors = 0

    for title, item_url, list_date in unique_items:
        try:
            # Check if already in DB
            cur.execute("SELECT id FROM gov_raw WHERE page_url=?", (item_url,))
            if cur.fetchone():
                skipped += 1
                continue

            html = fetch(item_url)
            det_title, det_date, source, content, attachments = parse_detail(html, item_url)

            final_title = det_title or title
            final_date = det_date or list_date

            if not final_title or not final_date:
                errors += 1
                print(f"  SKIP (no title/date): {item_url}", file=sys.stderr)
                continue

            date_rank = int(final_date.replace("-", ""))

            # Build summary from content
            content_text = content[:50000] if content else ""
            summary = re.sub(r'\s+', ' ', content_text[:200]).strip() or final_title

            attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""

            cur.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (page_url, title, publish_date, site_name, group_name, summary, content, attachments, source_url)
                   VALUES (?,?,?,?,?,?,?,?,?)""",
                (item_url, final_title, final_date,
                 SITE_NAME, GROUP, summary, content_text,
                 attachments_json, item_url)
            )
            if cur.rowcount > 0:
                conn.commit()
                saved += 1
                print(f"  OK: {final_title[:50]} | {final_date}", file=sys.stderr)
            else:
                skipped += 1

        except Exception as e:
            conn.rollback()
            errors += 1
            print(f"  ERR {item_url}: {e}", file=sys.stderr)

    conn.close()
    print(f"\n=== Done: saved={saved}, skipped={skipped}, errors={errors} ===", file=sys.stderr)


if __name__ == "__main__":
    main()
