#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
合浦县人民政府 - 乡镇部门 爬虫
http://www.hepu.gov.cn/ywdt/xzbm/

站点: 广西合浦县政府门户 (TRS 风格 / jsq/fenye.js 分页)
列表: <li><a href="./t{id}.shtml" title="完整标题">截断标题</a><span>YYYY-MM-DD</span></li>
分页: createPageHTML(50,0,"index","shtml","1849") -> index_{N}.shtml (N>=1, 首页无后缀)
详情: meta ArticleTitle/PubDate/ContentSource + div.trs_editor_view (TRS_UEDITOR Word转换, span逐字)
附件: 正文内嵌 ./W0{...}.doc 相对链接 -> 绝对化

输出约定: 所有日志走 stderr; stdout 仅输出 "新增: N" (调度器解析)
"""
import os
import re
import sys
import time
import sqlite3
import requests
import html as html_lib
import urllib.parse
from bs4 import BeautifulSoup
from datetime import datetime, timezone, timedelta
from urllib.parse import urljoin

# ---- 自动分页参数: --pages=N 或 --pages N 或 裸数字 ----
_MAX_PG = None
args = sys.argv[1:]
for i, a in enumerate(args):
    if a == "--pages" and i + 1 < len(args) and args[i + 1].isdigit():
        _MAX_PG = int(args[i + 1])
    elif a.startswith("--pages=") and a.split("=", 1)[1].isdigit():
        _MAX_PG = int(a.split("=", 1)[1])
    elif a.isdigit():
        _MAX_PG = int(a)
if _MAX_PG is not None:
    print(f"[AutoPg] max_pages={_MAX_PG}", file=sys.stderr)
# ---- END AUTO PAGES ----

BASE_URL = "http://www.hepu.gov.cn/ywdt/xzbm/"
DOMAIN = "www.hepu.gov.cn"
SITE_NAME = "合浦县乡镇部门"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 40
INCREMENTAL = "--incremental" in sys.argv
YEAR_LIMIT = 3

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

conn = sqlite3.connect(DB_PATH, timeout=60)
c = conn.cursor()

cutoff_date = datetime.now(timezone(timedelta(hours=8))).replace(tzinfo=None) - timedelta(days=365 * YEAR_LIMIT)


def log(msg):
    print(msg, file=sys.stderr)


def parse_list_page(html):
    """<li><a href="./t27953761.shtml" target="_blank" title="完整标题">截断</a><span>2026-07-27</span></li>"""
    items = []
    for m in re.finditer(
        r'<li><a href="\./(t\d+\.shtml)"[^>]*title="([^"]*)"[^>]*>.*?</a><span>(\d{4}-\d{2}-\d{2})</span></li>',
        html, re.S,
    ):
        items.append((m.group(2).strip(), m.group(1).strip(), m.group(3)))
    return items


def clean_content(content, page_url):
    """Convert trs_editor_view inner HTML to stored format (jlcity 同款):
    - keep <table> HTML intact
    - \\n\\n between <p> blocks
    - attachment <a> links embedded with absolute URL
    - strip <script>/<style>/span/font noise
    - dedupe paragraphs
    """
    if not content:
        return ""
    # strip outer wrapper divs keeping inner content
    content = re.sub(r'^\s*<div\s+class=["\'][^"\']*(?:detail-content|TRS_Editor|trs_editor_view)[^"\']*["\'][^>]*>', "", content, flags=re.I)
    content = re.sub(r"</div>\s*$", "", content)
    # noise cleanup
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    parts = []
    attachments = []  # (abs_url, name)
    has_table = 0

    # collect attachment links first (before any decompose)
    attach_links = []
    soup = BeautifulSoup(content, "html.parser")
    for a in soup.find_all("a"):
        href = a.get("href", "")
        if not href or href.startswith("#") or href.startswith("javascript:"):
            continue
        is_attach = bool(a.get("appendix")) or bool(re.search(r"\.(docx?|pdf|xlsx?|pptx?|et|ofd|wps|rar|zip|txt)$", href, re.I))
        if is_attach:
            txt = a.get_text(strip=True)
            if not txt:
                txt = os.path.basename(href.split("?")[0]) or "附件"
            txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
            abs_url = urllib.parse.urljoin(page_url, href)
            attach_links.append((a, abs_url, txt))

    # protect tables
    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"
    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content, flags=re.S | re.I)
    if re.search(r"<table[^>]*>", content, re.I):
        has_table = 1

    # protect attachment <a> tags
    link_protect = {}
    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
        return key
    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)

    # split into blocks by </p>
    # remove inline style attributes safely via BS4 (table/link are protected tokens now)
    _soup = BeautifulSoup(content2, "html.parser")
    # unwrap Word-conversion inline tags first
    for _sp in _soup.find_all(["span", "font"]):
        _sp.unwrap()
    for _tag in _soup.find_all(True):
        if "style" in _tag.attrs:
            del _tag.attrs["style"]
        # keep only structural attrs on a/img
        if _tag.name == "a":
            for _k in list(_tag.attrs):
                if _k not in ("href", "target"):
                    del _tag.attrs[_k]
        elif _tag.name == "img":
            for _k in list(_tag.attrs):
                if _k not in ("src", "alt"):
                    del _tag.attrs[_k]
        elif _tag.name == "table":
            for _k in list(_tag.attrs):
                if _k not in ("border", "cellspacing", "cellpadding"):
                    del _tag.attrs[_k]
        elif _tag.name in ("tr", "td", "th", "p", "div", "ul", "ol", "li", "br", "h1", "h2", "h3", "strong", "b"):
            _tag.attrs = {}
    content2 = str(_soup)
    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)

    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        # empty after tag strip?
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        # restore protected tokens
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)

    # remove any leftover placeholder tokens
    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)

    # final entity unescape
    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    # dedupe paragraphs
    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    # if no attachment links captured but content had some, append
    if attach_links:
        appended = []
        for a, abs_url, txt in attach_links:
            if abs_url not in out:
                appended.append(f'<p><a href="{abs_url}" target="_blank">{txt}</a></p>')
        if appended:
            out = (out + "\n\n" + "\n\n".join(appended)).strip()

    return out


def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        log(f"  [ERROR] fetch {url}: {e}")
        return None, None, None

    soup = BeautifulSoup(html, "html.parser")

    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    title = meta_title.get("content", "").strip() if meta_title else ""
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        h2 = soup.find("h2")
        if h2:
            title = h2.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True)

    # strip &middot; &nbsp; 实体前缀
    title = title.replace("&middot;", "").replace("&nbsp;", "").strip()
    title = re.sub(r"^[\s\u00a0·]+", "", title).strip()

    date_str = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date:
        content = meta_date.get("content", "")
        dm = re.match(r"(\d{4}-\d{2}-\d{2})", content)
        if dm:
            date_str = dm.group(1)

    content_html = ""
    article_con = soup.find("div", class_="article-con")
    if article_con:
        trs_div = article_con.find("div", class_=lambda c: c and "trs_editor_view" in c)
        if trs_div:
            content_html = clean_content(str(trs_div), url)
        else:
            content_html = clean_content(str(article_con), url)

    if not content_html or content_html == "<p>内容加载失败</p>":
        content_html = "<p>内容加载失败</p>"

    return title or "", content_html, date_str or ""


def article_exists(page_url):
    c.execute("SELECT COUNT(*) FROM gov_raw WHERE page_url = ?", (page_url,))
    return c.fetchone()[0] > 0


def sync_fts(row_id, title, site_name):
    try:
        # 2026-09-22: 先提交 gov_raw —— 下面手动写 FTS 会因触发器已写过同一
        #   rowid 而 IntegrityError，若不先 commit，这条记录会被一并回滚（静默丢数据）
        conn.commit()
        c.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                  (row_id, title, site_name, ""))
    except Exception as e:
        log(f"  [FTS ERROR] {e}")


def insert_article(title, page_url, content_html, date_str):
    display_url = page_url.replace("http://", "").replace("https://", "")
    summary = ""

    c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
    row = c.fetchone()
    existing_id = row[0] if row else None

    try:
        if existing_id:
            c.execute("""UPDATE gov_raw SET title=?, content=?, publish_date=?, source_url=?
                         WHERE id=?""",
                      (title, content_html, date_str, display_url, existing_id))
            row_id = existing_id
        else:
            c.execute("""INSERT INTO gov_raw
                (page_url, title, content, publish_date, site_name, source_url, summary, category)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
                (page_url, title, content_html, date_str, SITE_NAME, display_url, summary, ""))
            row_id = c.lastrowid
        if row_id:
            sync_fts(row_id, title, SITE_NAME)
        return True, existing_id is not None
    except Exception as e:
        log(f"  [DB ERROR] {e}")
        return False, False


def get_page_url(page_num):
    if page_num == 1:
        return BASE_URL
    return f"http://www.hepu.gov.cn/ywdt/xzbm/index_{page_num}.shtml"


def main():
    total_new = 0
    total_updated = 0
    total_skipped = 0
    total_failed = 0

    pages_to_crawl = min(MAX_PAGES, _MAX_PG) if _MAX_PG else MAX_PAGES
    log(f"  Pages to crawl: 1-{pages_to_crawl}")

    for page in range(1, pages_to_crawl + 1):
        url = get_page_url(page)
        log(f"  List page {page}/{pages_to_crawl}")

        r = None
        for attempt in range(3):
            try:
                r = requests.get(url, headers=HEADERS, timeout=20)
                r.encoding = "utf-8"
                break
            except Exception as e:
                log(f"    Retry {attempt+1}: {e}")
                time.sleep(2)

        if r is None:
            log(f"  [SKIP] page {page} failed")
            continue

        items = parse_list_page(r.text)
        if not items:
            log(f"    No items found")
            # 首页无条目 = 到尽头
            if page > 1:
                break
            continue

        log(f"    Found {len(items)} items")

        for title, item_url, list_date in items:
            full_url = urljoin(BASE_URL, item_url)

            if list_date:
                try:
                    item_date = datetime.strptime(list_date, "%Y-%m-%d")
                    if item_date < cutoff_date:
                        total_skipped += 1
                        continue
                except:
                    pass

            exists = article_exists(full_url)
            if exists:
                total_skipped += 1
                continue

            log(f"    Fetching: {title[:35]}...")
            d_title, content_html, detail_date = fetch_detail(full_url)

            if not d_title:
                d_title = title
            if not detail_date and list_date:
                detail_date = list_date

            ok, was_update = insert_article(d_title, full_url, content_html, detail_date)
            if ok:
                if was_update:
                    total_updated += 1
                else:
                    total_new += 1
            else:
                total_failed += 1

            time.sleep(0.3)

        conn.commit()
        log(f"  [COMMIT] page {page} done")

    conn.close()
    log(f"  完成: 新增 {total_new}, 更新 {total_updated}, 跳过 {total_skipped}, 失败 {total_failed}")
    # 调度器解析: stdout 仅一行 新增: N
    print(f"新增: {total_new}")


if __name__ == "__main__":
    main()
