#!/usr/bin/env python3
"""岢岚县人民政府 - 通知公告"""
import json, re, sys, requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
MAX_PAGES_DEFAULT = 150  # ~140 pages

# 支持 --pages 参数
import argparse as _AP
_AP_PARSER = _AP.ArgumentParser()
_AP_PARSER.add_argument("--pages", type=int, default=0, help="限制页数")
_AP_ARGS, _ = _AP_PARSER.parse_known_args()
MAX_PAGES = _AP_ARGS.pages if _AP_ARGS.pages > 0 else MAX_PAGES_DEFAULT

DB_PATH = "/root/search.db"
SITE_NAME = "岢岚县-通知公告"
CATEGORY = "通知公告"
GROUP = "岢岚县"
BASE_URL = "http://www.xzkl.gov.cn/zwyw/tzgg"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

SQL = """INSERT OR IGNORE INTO gov_raw
    (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)"""


def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def decode_resp(r):
    raw = r.content
    raw_lower = raw[:2000].lower()
    if b"gb2312" in raw_lower or b"gbk" in raw_lower or b"charset=gb" in raw_lower:
        return raw.decode("gbk", errors="replace")
    ct = r.headers.get("Content-Type", "")
    if "gb2312" in ct.lower() or "gbk" in ct.lower():
        return raw.decode("gbk", errors="replace")
    return raw.decode("utf-8", errors="replace")


def table_to_md(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def extract_content(content_div, detail_url):
    """Extract content from TRS_Editor or generic content div."""
    parts = []
    processed = set()
    for el in content_div.find_all(["p", "table", "img", "h1", "h2", "h3", "h4"], recursive=True):
        if id(el) in processed:
            continue
        processed.add(id(el))
        if el.name == "table":
            md = table_to_md(el)
            if md:
                parts.append(md)
        elif el.name == "img":
            src = el.get("src", "")
            if src:
                full_src = urljoin(detail_url, src) if not src.startswith("http") else src
                parts.append(f"![]({full_src})")
        elif el.name == "p":
            txt = el.get_text(" ", strip=True)
            if txt:
                txt = re.sub(r'[^\S\n]+', ' ', txt)
                parts.append(txt)
        else:
            txt = el.get_text(" ", strip=True)
            if txt:
                parts.append(txt)
    return "\n\n".join(parts)


def extract_attachments(soup, detail_url):
    attachments = []
    for a_tag in soup.find_all("a", href=True):
        href = a_tag["href"].strip().lower()
        if href.endswith(".pdf") or href.endswith(".doc") or href.endswith(".docx") or href.endswith(".xls") or href.endswith(".xlsx") or href.endswith(".rar") or href.endswith(".zip"):
            name = a_tag.get_text(strip=True) or href.split("/")[-1]
            full_url = urljoin(detail_url, a_tag["href"]) if not a_tag["href"].startswith("http") else a_tag["href"]
            # Avoid duplicates
            if not any(a["url"] == full_url for a in attachments):
                attachments.append({"name": name, "url": full_url})
    return attachments


def extract_list_page(page_num):
    """page_num: 1-based. index.html = 1, index_1.html = 2, etc."""
    if page_num == 1:
        url = f"{BASE_URL}/index.html"
    else:
        url = f"{BASE_URL}/index_{page_num-1}.html"

    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.raise_for_status()
    except Exception as e:
        print(f"  [ERR] page {page_num}: {e}", file=sys.stderr)
        return []

    html = decode_resp(r)
    soup = BeautifulSoup(html, "html.parser")
    items = []

    aon = soup.find("div", class_="aon-con")
    if not aon:
        print(f"  [WARN] page {page_num}: no aon-con found", file=sys.stderr)
        return []

    ul = aon.find("ul")
    if not ul:
        return []

    for li in ul.find_all("li", recursive=False):
        a_tag = li.find("a")
        if not a_tag or not a_tag.get("href"):
            continue
        href = a_tag["href"].strip()
        title = a_tag.get("title", "") or a_tag.get_text(strip=True)
        if not title:
            continue

        date_span = li.find("span")
        date_str = date_span.get_text(strip=True) if date_span else ""

        detail_url = urljoin(BASE_URL + "/", href)
        items.append({"title": title, "url": detail_url, "date": date_str})

    print(f"  Page {page_num}: {len(items)} items")
    return items


def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.raise_for_status()
    except Exception:
        return None, None, None, []

    html = decode_resp(r)
    soup = BeautifulSoup(html, "html.parser")

    # Title: h1 or title tag
    h1 = soup.find("h1")
    title = h1.get_text(strip=True) if h1 else ""
    if not title:
        mt = soup.find("title")
        if mt:
            title = mt.get_text(strip=True)
            # Clean breadcrumbs
            title = re.sub(r'[-_|]\s*通知公告.*$', '', title)
            title = re.sub(r'[-_|]\s*岢岚县人民政府.*$', '', title)

    # Date: find "发布时间" or YYYY-MM-DD pattern
    date_str = ""
    body_text = soup.get_text(" ", strip=True)
    m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', body_text)
    if m:
        date_str = m.group(1)
    if not date_str:
        m = re.search(r'日\s*期[：:]\s*(\d{4}-\d{2}-\d{2})', body_text)
        if m:
            date_str = m.group(1)
    if not date_str:
        m = re.search(r'(\d{4}-\d{2}-\d{2})\s', body_text)
        if m:
            date_str = m.group(1)

    # Content: TRS_Editor or div with content
    content = ""
    content_div = soup.find("div", class_="TRS_Editor") or \
                  soup.find("div", class_="content") or \
                  soup.find("div", class_="news-content") or \
                  soup.find("div", class_="article-content")
    if content_div:
        content = extract_content(content_div, url)
    else:
        # Fallback: use body
        body = soup.find("body")
        if body:
            # Remove header/nav/footer
            for tag in body.find_all(["script", "style", "nav", "header", "footer"]):
                tag.decompose()
            # Find main content area
            for cls in ["TRS_Editor", "content", "zoom", "article", "news"]:
                div = body.find("div", class_=lambda c: c and cls in c)
                if div:
                    content = extract_content(div, url)
                    break
            if not content:
                # Just get text from body
                txt = body.get_text(" ", strip=True)
                if len(txt) > 50:
                    content = txt

    attachments = extract_attachments(soup, url)

    # Empty content fallback
    if len(content.strip()) < 20:
        fallback = f"[{title or url.split('/')[-1]}]({url})"
        if attachments:
            for a in attachments:
                fallback += f"\n\n附件：[{a['name']}]({a['url']})"
        content = fallback

    return title, date_str, content, attachments


def crawl(test_mode=False, max_pages=MAX_PAGES):
    conn = init_db()
    cur = conn.cursor()
    total = 0
    errors = 0
    total_listed = 0

    for pn in range(1, max_pages + 1):
        items = extract_list_page(pn)
        total_listed += len(items)

        if not items:
            break

        for item in items:
            url = item["url"]
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                continue

            title, date_str, content, attachments = extract_detail(url)
            if not title:
                errors += 1
                continue

            final_title = title or item["title"]
            final_date = date_str or item["date"]
            att_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
            summary = (content[:200] if content else final_title).strip()
            if not summary:
                summary = final_title

            try:
                cur.execute(SQL, (
                    SITE_NAME, url, url, final_title.strip(), final_date,
                    summary, content, CATEGORY, att_json, GROUP,
                ))
                conn.commit()
                if cur.rowcount > 0:
                    total += 1
            except Exception as e:
                print(f"  DB error: {e}", file=sys.stderr)
                errors += 1

            if test_mode and total >= 5:
                break

        if test_mode and total >= 5:
            break

    conn.close()
    print(f"\n[DONE] {SITE_NAME}")
    print(f"  New: {total}, Errors: {errors}, Listed total: {total_listed}")
    return total


if __name__ == "__main__":
    mode = "test" if "--test" in sys.argv else "full"
    max_p = 5 if mode == "test" else MAX_PAGES
    if len(sys.argv) > 1 and sys.argv[1].isdigit():
        max_p = int(sys.argv[1])
    crawl(test_mode=(mode == "test"), max_pages=max_p)
