#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
乐平市人民政府 - 生态环境局 - 法定主动公开内容 - 行政许可类服务 - 清单管理
https://www.lps.gov.cn/gbmxxgkml/sthjj_15311/fdzdgknr_15651/xzglyfw/qdgl/

CMS: TRS（静态列表 + createPageHTML 分页, 模板同源乐平/景德镇系）
列表: ul.info-list > li.td01 > a[title=完整标题] + span 日期 (14条/页)
      首页 index.shtml；第 N 页(N>=2) = index_{N-1}.shtml (createPageHTML 第1参数=总页数)
详情: meta ArticleTitle/PubDate/ContentSource 齐全
      正文 div.trs_editor_view.TRS_UEDITOR (Word转换 <p><span> 逐段 + table 保留)
      附件正文内嵌 <a href="./downfile.jsp?..."> (urljoin(详情URL) 绝对化)
      尾部噪声: content5(空)/content6(信息来源)/content7(打印) 不在正文容器内, 无需截断
"""
import sys
import os
import re
import time
import html as html_lib
import urllib.parse
import sqlite3

# crawler_lib 所在目录 (classify_industry 行业分类复用)
for _p in ("/root/gov_crawler", "/root/gov_crawler"):
    if os.path.isdir(_p) and _p not in sys.path:
        sys.path.insert(0, _p)

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

# ---------------- config ----------------
BASE_URL = "https://www.lps.gov.cn"
LIST_PATH = "/gbmxxgkml/sthjj_15311/fdzdgknr_15651/xzglyfw/qdgl/"
LIST_URL = BASE_URL + LIST_PATH
SITE_NAME = "乐平市政府-生态环境局-清单管理"
GROUP_NAME = "江西"
SCRIPT_NAME = "crawl_lps_qdgl.py"
DB_PATH = "/mnt/data/search.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

# default: daily incremental = 1 page
_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass

_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1


# ---------------- helpers ----------------
def clean_title(t):
    """unescape entities + strip &middot;&nbsp; prefix + strip zero-width + collapse whitespace."""
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = re.sub(r"^\s*[·•]+\s*", "", t)  # &middot; 实体前缀
    t = t.replace("\xa0", " ").replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch(url, session=None, timeout=30, retries=3):
    import requests
    last = None
    for i in range(retries):
        try:
            r = (session or requests).get(url, headers=HEADERS, timeout=timeout)
            if r.status_code == 200:
                # lps 站 Content-Type 无 charset -> requests 默认 ISO-8859-1 导致中文乱码
                # apparent_encoding 检测 utf-8 正确, 显式设置
                r.encoding = r.apparent_encoding or "utf-8"
                return r
            last = r
        except Exception as e:
            last = e
        time.sleep(1.5 * (i + 1))
    if isinstance(last, Exception):
        raise last
    return last


def html_to_text(content, page_url):
    """Convert article-body inner HTML to stored format:
    - keep <table> HTML intact
    - \n\n between <p> blocks
    - attachment <a> links embedded with absolute URL
    - strip <script>/<style>/noise
    """
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)
    # strip style/class attributes via BS4 BEFORE paragraph splitting
    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for tag in _soup.find_all(True):
                for attr in ("style", "class", "lang", "dir", "align", "valign", "width", "height", "border", "cellpadding", "cellspacing"):
                    tag.attrs.pop(attr, None)
            content = str(_soup)
        except Exception:
            pass

    # 1) FIRST protect attachment <a> links (before table protection)
    content2 = content
    # remove attachment-file-type icon imgs (excel/word/pdf icons)
    content2 = re.sub(
        r'<img[^>]*src="[^"]*(?:fileTypeImages|icon)[^"]*\.(?:gif|png)"[^>]*/?>', "", content2,
        flags=re.I,
    )
    attachments = []
    link_protect = {}
    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = urllib.parse.urljoin(page_url, html_lib.unescape(href))
        attachments.append((abs_url, txt))
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
        return key
    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)

    # 2) THEN protect tables
    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"
    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content2, flags=re.S | re.I)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0

    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)
    content2 = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content2, flags=re.I)

    parts = []
    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)

    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)

    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    # dedupe paragraphs
    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    return out, has_table, attachments


def parse_list(html_text, page_url):
    """Extract (url, title, date) from list page."""
    items = []
    if BeautifulSoup:
        soup = BeautifulSoup(html_text, "html.parser")
        # precise sub-column: only within ul.info-list
        ul = soup.find("ul", class_=re.compile(r"info-list"))
        if ul:
            for li in ul.find_all("li"):
                a = li.find("a", href=True)
                if not a:
                    continue
                href = a.get("href", "").strip()
                if not re.search(r"t\d+\.shtml", href):
                    continue
                abs_url = urllib.parse.urljoin(page_url, href)
                title = clean_title(a.get("title", "") or a.get_text(" ", strip=True))
                span = li.find("span")
                date = span.get_text(strip=True) if span else ""
                if not re.match(r"^20\d{2}-\d{2}-\d{2}$", date):
                    date = ""
                items.append((abs_url, title, date))
    if not items:
        # regex fallback: li > a[title] + span
        for m in re.finditer(
            r'<li[^>]*>\s*<a[^>]+href="([^"]*?t\d+\.shtml)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<span[^>]*>([^<]*)</span>',
            html_text, re.S,
        ):
            abs_url = urllib.parse.urljoin(page_url, m.group(1))
            items.append((abs_url, clean_title(m.group(2)), m.group(3).strip()))
    return items


def parse_detail(html_text, page_url):
    """Extract title, date, content from detail page."""
    title = ""
    date = ""
    content_html = ""
    if BeautifulSoup:
        soup = BeautifulSoup(html_text, "html.parser")
        # title: meta ArticleTitle first, then h1
        mt = soup.find("meta", attrs={"name": re.compile(r"ArticleTitle", re.I)})
        if mt and mt.get("content"):
            title = clean_title(mt["content"])
        h1 = soup.find("h1")
        if not title and h1:
            title = clean_title(h1.get_text(" ", strip=True))
        # date: meta PubDate
        md = soup.find("meta", attrs={"name": re.compile(r"PubDate|Pubdate", re.I)})
        if md and md.get("content"):
            d = md["content"].strip()
            dm = re.match(r"(\d{4})-(\d{2})-(\d{2})", d)
            if dm:
                date = dm.group(0)
        # content: div.trs_editor_view.TRS_UEDITOR
        view = soup.find("div", class_=re.compile(r"trs_editor_view"))
        if view:
            content_html = str(view)
    if not title or not content_html:
        # regex fallback
        m = re.search(r'<meta\s+name="\s*ArticleTitle\s*"\s+content="([^"]*)"', html_text, re.I)
        if m:
            title = clean_title(m.group(1))
        m = re.search(r'<meta\s+name="\s*PubDate\s*"\s+content="(\d{4}-\d{2}-\d{2})"', html_text, re.I)
        if m:
            date = m.group(1)
        m = re.search(r'<div[^>]*class="[^"]*trs_editor_view[^"]*"[^>]*>(.*?)</div>', html_text, re.S)
        if m:
            content_html = m.group(1)
    return title, date, content_html


def store_record(conn, url, title, content, date, has_table):
    cur = conn.execute(
        "SELECT id FROM gov_raw WHERE page_url=?",
        (url,),
    )
    if cur.fetchone():
        return False
    try:
        # industry 分类: 复用 crawler_lib.classify_industry (小写匹配 + 英文词补齐)
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = html_lib.unescape(summary)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table)
        raise


def main():
    import requests
    s = requests.Session()
    s.headers.update(HEADERS)
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    conn.execute("PRAGMA journal_mode=WAL")

    new_count = 0
    skip_count = 0
    seen_urls = set()

    # 首页
    list_url = LIST_URL + "index.shtml"
    r = fetch(list_url, s)
    if not r:
        print("新增: 0")
        return
    html_text = r.text
    # 总页数: createPageHTML(总页数, 当前页, "index", "shtml", ...)
    total_pages = 1
    m = re.search(r"createPageHTML\((\d+)", html_text)
    if m:
        total_pages = int(m.group(1))

    for page in range(1, min(_PAGES, total_pages) + 1):
        if page > 1:
            # 第2页起: index_{page-1}.shtml
            page_url = f"{LIST_URL}index_{page-1}.shtml"
            r = fetch(page_url, s)
            if not r:
                continue
            html_text = r.text
        else:
            page_url = list_url
        items = parse_list(html_text, page_url)
        if not items:
            continue
        for url, title, date in items:
            if url in seen_urls:
                continue
            seen_urls.add(url)
            # detail
            try:
                dr = fetch(url, s)
            except Exception as e:
                print(f"  [ERR] {url}: {e}")
                skip_count += 1
                continue
            if not dr:
                skip_count += 1
                continue
            d_title, d_date, d_html = parse_detail(dr.text, url)
            if not d_title:
                d_title = title
            if not d_date:
                d_date = date
            content, has_table, attachments = html_to_text(d_html, url)
            if not content:
                skip_count += 1
                continue
            if store_record(conn, url, d_title, content, d_date, has_table):
                new_count += 1
            else:
                skip_count += 1

    conn.commit()
    conn.close()
    print(f"新增: {new_count}")


if __name__ == "__main__":
    main()
