#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""铜仁市万山区人民政府 - 通知公告 (TRS IGI CMS)"""
import json, re, sys, time, requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "铜仁市万山区人民政府-通知公告"
CATEGORY = "贵州"
GROUP = "贵州"
BASE_URL = "http://www.trws.gov.cn/xwzx/tzgg/"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

INSERT_SQL = """INSERT OR IGNORE INTO gov_raw
    (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)"""


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def extract_list_items(html_text):
    """Extract (title, url, date) from list page HTML."""
    soup = BeautifulSoup(html_text, "html.parser")
    items = []
    ul = soup.find("ul", class_="NewsList")
    if not ul:
        return items
    for li in ul.find_all("li", recursive=False):
        a = li.find("a")
        if not a:
            continue
        title = (a.get("title") or a.get_text(strip=True) or "").strip()
        if not title:
            continue
        href = a.get("href", "").strip()
        if not href:
            continue
        full_url = href if href.startswith("http") else urljoin("http://www.trws.gov.cn", href)
        span = li.find("span")
        date = span.get_text(strip=True) if span else ""
        items.append({"title": title, "url": full_url, "date": date})
    return items


def extract_detail(url):
    """Extract title, date, content, attachments from a detail page."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None, None, None, None
    except Exception as e:
        print(f"  [ERROR] fetch detail failed: {e}", file=sys.stderr)
        return None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")

    # --- Title ---
    title_el = soup.find("div", class_="ArticleTitle")
    title = title_el.get_text(strip=True) if title_el else ""

    # --- Date from meta ---
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # --- Content ---
    content_parts = []
    attachments = []
    ac = soup.find("div", class_="Article_Con")
    if ac:
        # Look for TRS_Editor inside
        trs = ac.find("div", class_=re.compile(r"TRS", re.I))
        content_div = trs if trs else ac

        # Remove noise
        for noise in content_div.find_all(["script", "style"]):
            noise.decompose()
        for cls_name in ["auto ewm", "video", "ewm"]:
            for el in content_div.find_all("div", class_=cls_name):
                el.decompose()

        # Extract p and table
        for child in content_div.find_all(["p", "table"], recursive=True):
            if child.name == 'table':
                tbl_html = html_table_to_html(child, url)
                if tbl_html:
                    content_parts.append(tbl_html)
            elif child.name == "p":
                txt = child.get_text(" ", strip=True)
                if txt and txt != "扫一扫在手机打开当前页面":
                    content_parts.append(txt)

        # --- Attachments ---
        for a in content_div.find_all("a", href=True):
            href = a["href"].strip()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                full_href = href if href.startswith("http") else urljoin(url, href)
                attachments.append({
                    "name": a.get_text(strip=True) or href.split("/")[-1],
                    "url": full_href,
                })

    if not content_parts:
        content_parts.append("[%s](%s)" % (title or url.split("/")[-1], url))

    content_text = "\n\n".join(content_parts)
    summary = content_text[:200] if content_text else title

    return title, pub_date, content_text, json.dumps(attachments, ensure_ascii=False)


def crawl(test_mode=False, max_pages=MAX_PAGES):
    conn = init_db()
    cur = conn.cursor()
    total = 0
    errors = 0

    for pn in range(1, max_pages + 1):
        if pn == 1:
            list_url = BASE_URL
        else:
            list_url = "http://www.trws.gov.cn/xwzx/tzgg/index_%d.html" % (pn - 1)

        print(f"[Page {pn}] {list_url}", file=sys.stderr)

        try:
            r = requests.get(list_url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  [ERROR] list page {pn} HTTP {r.status_code}", file=sys.stderr)
                errors += 1
                continue
        except Exception as e:
            print(f"  [ERROR] list page {pn}: {e}", file=sys.stderr)
            errors += 1
            continue

        items = extract_list_items(r.text)
        print(f"  Found {len(items)} items", file=sys.stderr)

        if not items:
            print(f"  [WARN] No items found, stopping", file=sys.stderr)
            break

        for item in items:
            # Check if already exists in DB
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
            if cur.fetchone():
                print(f"  SKIP (exists): {item['title'][:40]}", file=sys.stderr)
                continue

            title, pub_date, content, attachments = extract_detail(item["url"])
            if title is None:
                print(f"  [ERROR] detail unreachable: {item['url']}", file=sys.stderr)
                errors += 1
                continue

            final_title = title or item["title"]
            final_date = pub_date or item["date"]
            final_summary = content[:200] if content else final_title

            if test_mode:
                print(f"  [{final_date}] {final_title[:50]}", file=sys.stderr)
                total += 1
                continue

            try:
                cur.execute(INSERT_SQL, (
                    SITE_NAME, item["url"], item["url"],
                    final_title.strip(), final_date,
                    final_summary.strip(), content or ("[%s](%s)" % (final_title, item["url"])),
                    CATEGORY, attachments, GROUP,
                ))
                conn.commit()
                total += 1
                print(f"  OK: {final_title[:40]}", file=sys.stderr)
            except Exception as e:
                print(f"  DB error: {e}", file=sys.stderr)
                errors += 1

        time.sleep(1)

    conn.close()
    print(f"\n[DONE] Total: {total}, Errors: {errors}", file=sys.stderr)
    return total


if __name__ == "__main__":
    test_mode = "--test" in sys.argv
    mp = MAX_PAGES
    for i, a in enumerate(sys.argv):
        if a.startswith("--max-pages="):
            mp = int(a.split("=")[1])
    crawl(test_mode=test_mode, max_pages=mp)
