#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
Crawler for 仁寿县人民政府 - 环评审批
CMS: Visual SiteBuilder 9 (VSB)
List: Table tr[id^=line253142_] > td > a (title) + td (date), 17 items, 1 page
Detail: meta ArticleTitle, meta PubDate, div#vsb_content_4 content
Attachments: a[href*=pdf|doc|rar|zip]

Usage:
    python3 crawl_rsgov_hjsp.py             # Full crawl
"""

import requests, re, json, sqlite3, time, os, sys
from datetime import datetime
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "仁寿县人民政府-环评审批"
CATEGORY = "环评审批"
GROUP = "四川"
BASE_URL = "http://www.rs.gov.cn"
LIST_URL = "http://www.rs.gov.cn/zwgk/zfxxgk/fdzdgknr/zdxxgk/sthj1/hjsp.htm"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

# ---------- Content extraction ----------

SKIP_TEXTS = [
    "【打印本页】", "【关闭窗口】", "打印本页", "关闭窗口",
    "转载分享：", "浏览量：", "相关附件：", "相关稿件：",
    "扫一扫在手机打开", "附件：", "累计次数：", "字号：",
]


def render_paragraph(p_tag, detail_url):
    parts = []
    for elem in p_tag.children:
        if isinstance(elem, str):
            t = elem.strip()
            if t:
                parts.append(t)
        elif elem.name == "a":
            href = elem.get("href", "").strip()
            text = elem.get_text(strip=True)
            if href and text:
                full = urljoin(detail_url, href)
                parts.append(f"[{text}]({full})")
            elif text:
                parts.append(text)
        elif elem.name == "img":
            src = elem.get("src", "")
            if src:
                alt = elem.get("alt", "")
                full_src = urljoin(detail_url, src)
                parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
        elif elem.name == "br":
            parts.append("\n")
        elif hasattr(elem, "get_text"):
            t = elem.get_text(" ", strip=True)
            if t:
                parts.append(t)
    return " ".join(parts).strip()


def extract_images(content_div, detail_url):
    imgs = []
    for img in content_div.find_all("img"):
        src = img.get("src", "")
        if src:
            full_src = urljoin(detail_url, src)
            alt = img.get("alt", "")
            imgs.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
            img.decompose()
    return imgs


def extract_tables(content_div):
    tables = []
    for table in content_div.find_all("table"):
        table_html = str(table)
        if len(table.get_text(strip=True)) >= 15:
            tables.append(table_html)
        table.decompose()
    return tables


def table_to_md(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def extract_content(content_div, detail_url):
    parts = []

    for s in content_div.find_all(["script", "style"]):
        s.decompose()

    table_htmls = extract_tables(content_div)
    img_lines = extract_images(content_div, detail_url)

    for child in list(content_div.find_all(["p", "table", "div", "section", "center", "font"], recursive=True)):
        if child.name == "p":
            txt = render_paragraph(child, detail_url)
            if txt and not any(sk in txt for sk in SKIP_TEXTS):
                if not parts or txt not in parts[-1]:
                    parts.append(txt)

    result = "\n\n".join(parts)
    if table_htmls:
        for t_html in table_htmls:
            table_tag = BeautifulSoup(t_html, "html.parser").find("table")
            if table_tag:
                result += "\n\n" + table_to_md(table_tag)
    if img_lines:
        result += "\n\n" + "\n\n".join(img_lines)

    return result.strip()


# ---------- Attachments ----------

def extract_attachments(soup, detail_url):
    atts = []
    seen = set()
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|rar|zip|txt)$", href, re.I):
            full_url = urljoin(detail_url, href)
            name = a.get_text(strip=True) or href.split("/")[-1].split("?")[0]
            if full_url not in seen:
                seen.add(full_url)
                atts.append({"name": name, "url": full_url})
    return atts


# ---------- Detail ----------

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  fetch_detail error {url}: {e}", file=sys.stderr)
        return None, None, None, []

    soup = BeautifulSoup(r.text, "html.parser")

    # --- Title from meta ---
    title = ""
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()
    if not title:
        t = soup.find("title")
        if t:
            title = t.get_text(strip=True).replace("-仁寿县人民政府", "").strip()

    # --- Date from meta ---
    date_str = ""
    mt = soup.find("meta", attrs={"name": "PubDate"})
    if mt and mt.get("content"):
        date_str = mt["content"].strip()[:10]

    # --- Content ---
    content_div = soup.find("div", id="vsb_content_4")
    if not content_div:
        content_div = soup.find("div", class_="newscontent_s")
    if not content_div:
        # Fallback: try any large text area
        body = soup.find("body")
        if body:
            for div in body.find_all("div"):
                txt = div.get_text(strip=True)
                if len(txt) > 200:
                    content_div = div
                    break

    content = ""
    attachments = []

    if content_div:
        content = extract_content(content_div, url)
        attachments = extract_attachments(soup, url)
    else:
        content = f'<p><a href="{url}">{title or url}</a></p>'

    if not content or len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title or url}</a></p>'

    # Append attachments
    if attachments:
        for a in attachments:
            inline_marker = f"[{a['name']}]({a['url']})"
            if inline_marker not in content:
                content += f"\n\n附件：[{a['name']}]({a['url']})"

    return title, date_str, content, attachments


# ---------- List ----------

def extract_list_page():
    try:
        r = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  Error fetching list: {e}", file=sys.stderr)
        return []

    if r.status_code != 200:
        print(f"  List: HTTP {r.status_code}", file=sys.stderr)
        return []

    soup = BeautifulSoup(r.text, "html.parser")

    items = []
    for tr in soup.find_all("tr", id=re.compile(r"^line253142_")):
        tds = tr.find_all("td")
        if len(tds) < 2:
            continue
        a = tds[0].find("a")
        if not a or not a.get("href"):
            continue

        href = a["href"].strip()
        title = a.get("title") or a.get_text(strip=True)
        page_url = urljoin(LIST_URL, href)
        date_str = tds[1].get_text(strip=True)

        if title and page_url:
            items.append((title, page_url, date_str))

    return items


# ---------- DB ----------

def with_retry(fn, desc="DB op", max_attempts=10, delay=5):
    for attempt in range(1, max_attempts + 1):
        try:
            return fn()
        except sqlite3.OperationalError as e:
            if "locked" in str(e) and attempt < max_attempts:
                print(f"  {desc}: locked (attempt {attempt}/{max_attempts}), retry in {delay}s...", file=sys.stderr)
                time.sleep(delay)
            else:
                print(f"  {desc} failed: {e}", file=sys.stderr)
                return False
    return False


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn


def save_article(conn, title, page_url, publish_date, content, attachments):
    summary = content if content else title
    summary = re.sub(r"\s+", " ", summary).strip()
    att_json = json.dumps(attachments, ensure_ascii=False) if attachments else "[]"
    source_url = page_url
    try:
        cur = conn.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
               VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""",
            (SITE_NAME, source_url, page_url, title.strip(), publish_date,
             summary, content, CATEGORY, att_json, GROUP),
        )
        return cur.rowcount > 0
    except Exception as e:
        print(f"  DB error: {e}", file=sys.stderr)
        return False


# ---------- Main ----------

def main():
    print(f"Scraping {SITE_NAME}", file=sys.stderr)
    conn = init_db()
    total_new = 0

    # Clean old data (BOTH gov_raw and FTS index)
    def do_clean():
# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)         conn.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name = ?)", (SITE_NAME,))
# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)         conn.execute("DELETE FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
        conn.commit()
    with_retry(do_clean, desc=f"Clean {SITE_NAME}")
    print(f"  Cleaned old data for {SITE_NAME}", file=sys.stderr)

    articles = extract_list_page()
    print(f"  Found {len(articles)} articles on list page", file=sys.stderr)

    for idx, (title, page_url, date) in enumerate(articles):
        print(f"  [{idx+1}/{len(articles)}] Fetching: {title[:50]}...", file=sys.stderr)

        det_title, det_date, det_content, det_atts = fetch_detail(page_url)
        final_title = det_title or title
        final_date = det_date or date
        final_content = det_content or f"[{final_title}]({page_url})"
        final_atts = det_atts or []

        if save_article(conn, final_title, page_url, final_date, final_content, final_atts):
            total_new += 1

        time.sleep(0.5)

    conn.commit()

    # Rebuild FTS index for this site
    def do_fts():
# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)         conn.execute("DELETE FROM gov_search WHERE rowid IN (SELECT id FROM gov_raw WHERE site_name = ?)", (SITE_NAME,))
        conn.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) SELECT id, title, site_name, summary FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
        conn.commit()
    with_retry(do_fts, desc=f"FTS rebuild {SITE_NAME}")
    print(f"  FTS index rebuilt for {SITE_NAME}", file=sys.stderr)

    conn.close()
    print(f"\nDone! Total: {len(articles)}, New: {total_new}", file=sys.stderr)


if __name__ == "__main__":
    main()
