#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
吉林市人民政府 - 通知公告 (www.jlcity.gov.cn/zw/tzgg/)
CMS: TRS (meta ArticleTitle/PubDate/ContentSource, appendix 附件, WcmStatic 统计)
列表: div.home-list-info > a[title] (双 div: 标题+日期), 每页约10条
分页: index.html (第1页) / index_{N}.html (0-based, 共56页)
详情: div.detail-title 标题 + div.detail-info 日期 + div.detail-content > div.TRS_Editor 正文
      Word 转换 (p.MsoNormal + span 逐字) + <table> 表格保留 + appendix 附件内嵌
"""
import sys
import os
import re
import time
import html as html_lib
import urllib.parse
import sqlite3

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

# ---------------- config ----------------
BASE_URL = "http://www.jlcity.gov.cn"
LIST_PATH = "/zw/tzgg/"
LIST_URL = BASE_URL + LIST_PATH
SITE_NAME = "吉林市人民政府-通知公告"
GROUP_NAME = "吉林"
SCRIPT_NAME = "crawl_jlcity_tzgg.py"
DB_PATH = "/mnt/data/search.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

# default: daily incremental = 1 page
_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass

_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


# ---------------- helpers ----------------
def clean_title(t):
    """Strip &middot;&nbsp; entity prefixes + unescape all entities + strip zero-width."""
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch(url, session=None, timeout=30, retries=3):
    import requests
    last = None
    for i in range(retries):
        try:
            r = (session or requests).get(url, headers=HEADERS, timeout=timeout)
            if r.status_code == 200:
                return r
            last = r
        except Exception as e:
            last = e
        time.sleep(1.5 * (i + 1))
    if isinstance(last, Exception):
        raise last
    return last


def html_to_text(content, page_url):
    """Convert detail-content inner HTML to stored format:
    - keep <table> HTML intact
    - \n\n between <p> blocks
    - attachment <a> links embedded with absolute URL
    - strip <script>/<style>/noise
    """
    if not content:
        return "", 0, []
    # strip outer wrapper divs (detail-content / TRS_Editor) keeping inner content
    content = re.sub(r"^\s*<div\s+class=[\"'](?:detail-content|TRS_Editor)[\"'][^>]*>", "", content, flags=re.I)
    content = re.sub(r"</div>\s*$", "", content)
    # noise cleanup
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)

    soup = None
    parts = []
    attachments = []  # (abs_url, name)
    has_table = 0

    # collect attachment links first (before any decompose)
    attach_links = []
    if BeautifulSoup:
        soup = BeautifulSoup(content, "html.parser")
        for a in soup.find_all("a"):
            href = a.get("href", "")
            if not href or href.startswith("#") or href.startswith("javascript:"):
                continue
            is_attach = bool(a.get("appendix")) or bool(re.search(r"\.(docx?|pdf|xlsx?|pptx?|et|ofd|wps|rar|zip|txt)$", href, re.I))
            if is_attach:
                txt = a.get_text(strip=True)
                if not txt:
                    txt = os.path.basename(href.split("?")[0]) or "附件"
                txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
                abs_url = urllib.parse.urljoin(page_url, href)
                attach_links.append((a, abs_url, txt))

    # use regex-based extraction on raw HTML for table preservation
    # protect tables
    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"
    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content, flags=re.S | re.I)
    if re.search(r"<table[^>]*>", content, re.I):
        has_table = 1

    # protect attachment <a> tags
    link_protect = {}
    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, href)
        attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
        return key
    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)

    # split into blocks by </p>
    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)
    # strip Word-conversion inline tags (span/font) but keep their text
    content2 = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content2, flags=re.I)
    content2 = re.sub(r"</?o:p>", "", content2, flags=re.I)

    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        # empty after tag strip?
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        # restore protected tokens
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        # if block is pure table, keep html; else keep html too (contains links/tables possibly)
        b = re.sub(r"<a\s[^>]*href=\"([^\"]+)\"[^>]*>(.*?)</a>", lambda m: link_protect.get(m.group(0), m.group(0)), b)
        parts.append(b.strip())

    # restore leftover protected tokens if any
    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)

    # remove any leftover placeholder tokens
    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)

    # final entity unescape
    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    # dedupe paragraphs
    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    # if no attachment links captured but content had some, append
    if attach_links:
        appended = []
        for a, abs_url, txt in attach_links:
            if abs_url not in out:
                appended.append(f'<p><a href="{abs_url}" target="_blank">{txt}</a></p>')
        if appended:
            out = (out + "\n\n" + "\n\n".join(appended)).strip()

    return out, has_table, attachments


def parse_list(html_text, page_url):
    """Extract (url, title, date) from list page."""
    items = []
    if BeautifulSoup:
        soup = BeautifulSoup(html_text, "html.parser")
        box = soup.find("div", class_="home-list-info")
        if box:
            for a in box.find_all("a", href=True):
                href = a.get("href", "").strip()
                if not re.search(r"/20\d{4}/t\d+_\d+\.html$", href):
                    continue
                abs_url = urllib.parse.urljoin(LIST_URL, href)
                title = a.get("title") or a.get_text(" ", strip=True)
                title = clean_title(title)
                divs = a.find_all("div")
                date = ""
                for d in divs:
                    t = d.get_text(strip=True)
                    if re.match(r"^20\d{2}-\d{2}-\d{2}$", t):
                        date = t
                        break
                items.append((abs_url, title, date))
    if not items:
        # regex fallback
        for m in re.finditer(r'<a[^>]+href="([^"]*?/20\d{4}/t\d+_\d+\.html)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>', html_text, re.S):
            href, title, inner = m.group(1), m.group(2), m.group(3)
            abs_url = urllib.parse.urljoin(LIST_URL, href)
            date_m = re.search(r"20\d{2}-\d{2}-\d{2}", inner)
            items.append((abs_url, clean_title(title), date_m.group(0) if date_m else ""))
    return items


def parse_detail(html_text, page_url):
    """Extract title, date, content from detail page."""
    title = ""
    date = ""
    if BeautifulSoup:
        soup = BeautifulSoup(html_text, "html.parser")
        m = soup.find("meta", attrs={"name": "ArticleTitle"})
        if m and m.get("content"):
            title = clean_title(m["content"])
        m = soup.find("meta", attrs={"name": "PubDate"})
        if m and m.get("content"):
            date = m["content"].strip()[:10]
        dc = soup.find("div", class_="detail-content")
        if dc:
            content_html = str(dc)
            # strip the wzfj script block
            content_html = re.sub(r"<script[^>]*>.*?</script>", "", content_html, flags=re.S | re.I)
            # find inner TRS_Editor if present
            te = dc.find("div", class_="TRS_Editor")
            if te:
                content_html = str(te)
            if not title:
                dt = soup.find("div", class_="detail-title")
                if dt:
                    title = clean_title(dt.get_text(" ", strip=True))
            if not date:
                di = soup.find("div", class_="detail-info")
                if di:
                    dm = re.search(r"日期[:：]\s*(\d{4}-\d{2}-\d{2})", di.get_text())
                    if dm:
                        date = dm.group(1)
            return title, date, content_html
    # regex fallback
    m = re.search(r'name="ArticleTitle"\s+content="([^"]*)"', html_text)
    if m:
        title = clean_title(m.group(1))
    m = re.search(r'name="PubDate"\s+content="([^"]*)"', html_text)
    if m:
        date = m.group(1).strip()[:10]
    idx = html_text.find('class="detail-content"')
    if idx != -1:
        # find matching closing div by scanning
        seg = html_text[idx:]
        depth = 0
        end = -1
        for mm in re.finditer(r"<div\b[^>]*>|</div>", seg):
            if mm.group(0).startswith("</div>"):
                depth -= 1
                if depth == 0:
                    end = mm.end()
                    break
            else:
                depth += 1
        if end > 0:
            content_html = seg[:end]
            content_html = re.sub(r"<script[^>]*>.*?</script>", "", content_html, flags=re.S | re.I)
            te = re.search(r'<div class="TRS_Editor">', content_html)
            if te:
                content_html = content_html[te.start():]
            return title, date, content_html
    return title, date, ""


def store_record(conn, url, title, content, date, has_table):
    cur = conn.execute(
        "SELECT id FROM gov_raw WHERE page_url=?",
        (url,),
    )
    if cur.fetchone():
        return False
    try:
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, "other"),
        )
        if cur.rowcount > 0:
            # sync FTS
            summary = re.sub(r"<[^>]+>", "", content)
            summary = html_lib.unescape(summary)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table)
        raise


def main():
    import requests
    s = requests.Session()
    s.headers.update(HEADERS)
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    conn.execute("PRAGMA journal_mode=WAL")

    new_count = 0
    skip_count = 0
    seen_urls = set()

    for page in range(1, _PAGES + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = f"{LIST_URL}index_{page-1}.html"
        try:
            r = fetch(url, session=s)
            html_text = r.content.decode("utf-8", errors="replace")
        except Exception as e:
            print(f"Page {page} fetch error: {e}", flush=True)
            break
        items = parse_list(html_text, url)
        print(f"Page {page}: found {len(items)} items", flush=True)
        if not items and page > 1:
            print("Empty page, stop pagination", flush=True)
            break
        for abs_url, title, date in items:
            if abs_url in seen_urls:
                continue
            seen_urls.add(abs_url)
            try:
                r = fetch(abs_url, session=s)
                dhtml = r.content.decode("utf-8", errors="replace")
            except Exception as e:
                print(f"  detail error {abs_url}: {e}", flush=True)
                skip_count += 1
                continue
            dtitle, ddate, content_html = parse_detail(dhtml, abs_url)
            if not dtitle:
                dtitle = title
            if not ddate:
                ddate = date
            content, has_table, _ = html_to_text(content_html, abs_url)
            # empty content check
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            if plain_len < 10:
                print(f"  skip empty: {dtitle[:30]}", flush=True)
                skip_count += 1
                continue
            ok = store_record(conn, abs_url, dtitle, content, ddate, has_table)
            if ok:
                new_count += 1
                print(f"  + {dtitle[:40]} ({ddate})", flush=True)
            else:
                skip_count += 1
            time.sleep(0.3)
        time.sleep(0.5)

    conn.commit()
    conn.close()
    print(f"新增: {new_count}", flush=True)
    print(f"跳过: {skip_count}", flush=True)
    print(f"=== {SITE_NAME} done ===", flush=True)


if __name__ == "__main__":
    main()
