#!/usr/bin/env python3
"""湘潭天易经济开发区 - 政务公开"""
import json, re, sys, requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "湘潭天易经开区-政务公开"
CATEGORY = "政务公开"
BASE_URL = "http://www.xtx.gov.cn/10115/10116/10474/11535/index.htm"
OFFSET_URL = "http://www.xtx.gov.cn/xtx/10115/10116/10474/11535/index.jsp?pager.offset={offset}&pager.desc=false"
STEP = 15
MAX_OFFSET = 300

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

SQL = """INSERT OR IGNORE INTO gov_raw
    (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)"""


def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def decode_resp(r):
    raw = r.content
    ct = r.headers.get("Content-Type", "")
    if "gb2312" in ct.lower() or "gbk" in ct.lower():
        return raw.decode("gbk", errors="replace")
    raw_lower = raw[:2000].lower()
    if b"gb2312" in raw_lower or b"gbk" in raw_lower or b"charset=gb" in raw_lower:
        return raw.decode("gbk", errors="replace")
    return raw.decode("utf-8", errors="replace")


def table_to_md(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def clean_noise(xl_div):
    """Remove noise elements from the content div."""
    # Remove QR code container
    for div in xl_div.find_all("div", id=lambda x: x and "div_div" in str(x)):
        div.decompose()
    # Remove qr_container
    for div in xl_div.find_all("div", id="qr_container"):
        div.decompose()
    # Remove canvas elements
    for c in xl_div.find_all("canvas"):
        c.decompose()
    # Remove 相关解读 section
    for div in xl_div.find_all("div", class_="xiangguan"):
        div.decompose()
    # Remove empty spacer tables (single cell, height only)
    for table in xl_div.find_all("table"):
        rows = table.find_all("tr")
        if len(rows) == 1:
            cells = rows[0].find_all(["td", "th"])
            if len(cells) == 1 and not cells[0].get_text(strip=True):
                table.decompose()
    # Remove wbr tags
    for wbr in xl_div.find_all("wbr"):
        wbr.unwrap()
    # Remove script/style
    for tag in xl_div.find_all(["script", "style", "iframe"]):
        tag.decompose()


def extract_content(xl_div, detail_url):
    clean_noise(xl_div)
    parts = []
    processed = set()

    for el in xl_div.find_all(["p", "table", "img", "a", "h1", "h2", "h3", "h4"], recursive=True):
        if id(el) in processed:
            continue
        processed.add(id(el))

        if el.name == "table":
            md = table_to_md(el)
            if md:
                parts.append(md)
        elif el.name == "img":
            src = el.get("src", "")
            if src:
                alt = el.get("alt", "")
                full_src = urljoin(detail_url, src) if not src.startswith("http") else src
                parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
            for sub_img in el.find_all("img", recursive=True):
                processed.add(id(sub_img))
        elif el.name == "a" and re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', el.get("href", ""), re.I):
            href = el["href"].strip()
            name = el.get_text(strip=True) or href.split("/")[-1]
            full_url = urljoin(detail_url, href) if not href.startswith("http") else href
            parts.append(f"[{name}]({full_url})")
        elif el.name == "p":
            txt = el.get_text(" ", strip=True)
            if txt:
                skip_phrases = ["扫一扫", "相关解读", "详细内容：请", "详细内容：请下载附件"]
                if any(p in txt for p in skip_phrases):
                    continue
                txt = re.sub(r'[^\S\n]+', ' ', txt)
                parts.append(txt)
        else:
            txt = el.get_text(" ", strip=True)
            if txt:
                skip_phrases = ["扫一扫", "相关解读", "详细内容：请"]
                if any(p in txt for p in skip_phrases):
                    continue
                parts.append(txt)

    return "\n\n".join(parts)


def extract_attachments(soup, detail_url):
    attachments = []
    for a in soup.select('div.xl-xqnr a[href]'):
        href = a["href"].strip()
        if not re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            continue
        name = a.get_text(strip=True) or href.split("/")[-1]
        full_url = urljoin(detail_url, href) if not href.startswith("http") else href
        if not any(aa["url"] == full_url for aa in attachments):
            attachments.append({"name": name, "url": full_url})
    return attachments


def extract_list_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.raise_for_status()
    except Exception as e:
        print(f"  [ERR] {url}: {e}", file=sys.stderr)
        return []
    html = decode_resp(r)
    soup = BeautifulSoup(html, "html.parser")
    items = []

    table = soup.find("table", class_="table")
    if not table:
        tbody = soup.find("tbody")
        if not tbody:
            print(f"  [WARN] no table/tbody found", file=sys.stderr)
            return []
        rows = tbody.find_all("tr")
    else:
        tbody = table.find("tbody") or table
        rows = tbody.find_all("tr")

    for tr in rows:
        tds = tr.find_all("td")
        if len(tds) < 3:
            continue
        a_tag = tds[1].find("a", href=True)
        if not a_tag:
            continue
        href = a_tag["href"].strip()
        title = a_tag.get("title", "") or a_tag.get_text(strip=True)
        if not title:
            continue
        date_str = tds[2].get_text(strip=True)
        full_url = urljoin(url, href)
        items.append({"title": title, "url": full_url, "date": date_str})

    return items


def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        if r.status_code == 404:
            return None, None, None, []
        r.raise_for_status()
    except Exception as e:
        return None, None, None, []

    html = decode_resp(r)
    if "404" in html[:500] or "Page Not Found" in html:
        return None, None, None, []

    soup = BeautifulSoup(html, "html.parser")

    h2 = soup.find("h2")
    title = h2.get_text(strip=True) if h2 else ""

    date_str = ""
    h6 = soup.find("h6")
    if h6:
        spans = h6.find_all("span")
        for span in spans:
            txt = span.get_text(strip=True)
            m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
            if m:
                date_str = m.group(1)
                break

    xl = soup.find("div", class_="xl-xqnr")
    content = ""
    if xl:
        content = extract_content(xl, url)

    attachments = extract_attachments(soup, url)

    if len(content.strip()) < 20:
        fallback = f"[{title or url.split('/')[-1]}]({url})"
        if attachments:
            attach_links = [f"附件：[{a['name']}]({a['url']})" for a in attachments]
            fallback += "\n\n" + "\n\n".join(attach_links)
        content = fallback

    return title, date_str, content, attachments


def crawl(test_mode=False):
    conn = init_db()
    cur = conn.cursor()
    total = 0
    errors = 0
    total_listed = 0
    visited_urls = set()

    offsets = [0] + list(range(STEP, MAX_OFFSET + 1, STEP))

    for idx, offset in enumerate(offsets):
        if offset == 0:
            url = BASE_URL
        else:
            url = OFFSET_URL.format(offset=offset)
        items = extract_list_page(url)
        total_listed += len(items)
        print(f"  Page {idx + 1} (offset={offset}): {len(items)} items")

        if not items:
            break

        for item in items:
            u = item["url"]
            if u in visited_urls:
                continue
            visited_urls.add(u)

            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (u,))
            if cur.fetchone():
                continue

            title, date_str, content, attachments = extract_detail(u)
            if not title:
                errors += 1
                continue

            final_title = title or item["title"]
            final_date = date_str or item["date"]
            att_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
            summary = (content[:200] if content else final_title).strip()
            if not summary:
                summary = final_title

            try:
                cur.execute(SQL, (
                    SITE_NAME, u, u, final_title.strip(), final_date,
                    summary, content, CATEGORY, att_json, CATEGORY,
                ))
                conn.commit()
                if cur.rowcount > 0:
                    total += 1
            except Exception as e:
                print(f"  DB error: {e}", file=sys.stderr)
                errors += 1

            if test_mode and total >= 5:
                break
        if test_mode and total >= 5:
            break

    conn.close()
    print(f"\n[DONE] {SITE_NAME}")
    print(f"  New: {total}, Errors: {errors}, Listed total: {total_listed}")
    return total


if __name__ == "__main__":
    test = "--test" in sys.argv
    crawl(test_mode=test)
