#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
联合赤道环境评价股份有限公司 - 项目公示 (www.lheia.com/projectPublicity.html)
CMS: 自研 Java (Spring Boot + Vue, lheiaows.lheia.com/prod-api)
列表: POST https://lheiaows.lheia.com/prod-api/project/pub/ocw/list?pageNum=N&pageSize=50
      body {} 即可. 返回 data.list[]: projectName / projectInfo(HTML全文) / pubTime / fileList(附件)
      50条/页, 共311条/7页
详情: 列表即全文 (projectInfo) — 无需详情页; 详情页 project_details.html?id= 备用
附件: fileList[].path → http://fileIp/path (直接IP, 外链)
"""
import sys
import os
import re
import time
import html as html_lib
import json
import sqlite3

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

# ---------------- config ----------------
API_URL = "https://lheiaows.lheia.com/prod-api/project/pub/ocw/list"
DETAIL_TMPL = "https://www.lheia.com/project_details.html?id={}"
SITE_NAME = "联合赤道环境评价-项目公示"
GROUP_NAME = "企业"
SCRIPT_NAME = "crawl_lheia_pub.py"
DB_PATH = "/mnt/data/search.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Content-Type": "application/json",
}

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass

_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1
PAGE_SIZE = 50


def clean_title(t):
    if not t:
        return ""
    t = html_lib.unescape(t)
    t = t.replace("\u200b", "").replace("\u200c", "").replace("\u200d", "").replace("\ufeff", "")
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch_list(session, page):
    import requests
    r = session.post(f"{API_URL}?pageNum={page}&pageSize={PAGE_SIZE}",
                     data="{}", headers=HEADERS, timeout=30)
    if r.status_code != 200:
        raise RuntimeError(f"API HTTP {r.status_code}: {r.text[:200]}")
    data = r.json().get("data", {})
    total = data.get("total") or 0
    items = []
    for it in data.get("list", []):
        title = clean_title(it.get("projectName", ""))
        pub = (it.get("pubTime") or "").strip()
        dm = re.match(r"(\d{4})-(\d{2})-(\d{2})", pub)
        date = dm.group(0) if dm else ""
        content_html = it.get("projectInfo") or ""
        pid = str(it.get("id", ""))
        detail_url = DETAIL_TMPL.format(pid) if pid else ""
        # attachments from fileList
        file_list = it.get("fileList") or []
        attachments = []
        for f in file_list:
            if not isinstance(f, dict):
                continue
            fpath = f.get("path") or ""
            fip = f.get("fileIp") or ""
            fname = f.get("uploadName") or f.get("saveName") or os.path.basename(fpath)
            if fpath:
                url = f"http://{fip}{fpath}" if fip else fpath
                attachments.append((url, fname))
        if title and detail_url:
            items.append((detail_url, title, date, content_html, attachments))
    try:
        total = int(total)
    except Exception:
        total = 0
    return items, total


def html_to_text(content, page_url, attachments=None):
    if not content:
        return "", 0, []
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)
    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for tag in _soup.find_all(True):
                for attr in ("style", "class", "lang", "dir", "align", "valign", "width", "height", "border", "cellpadding", "cellspacing"):
                    tag.attrs.pop(attr, None)
            content = str(_soup)
        except Exception:
            pass

    content2 = content
    content2 = re.sub(
        r'<img[^>]*src="[^"]*fileTypeImages/icon_[a-z]+\.gif"[^>]*/?>', "", content2,
        flags=re.I,
    )
    link_protect = {}
    def _link_repl(m):
        href = m.group(1)
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        txt = re.sub(r"^附件[:：]\s*", "", txt).strip()
        if not txt:
            txt = os.path.basename(href.split("?")[0]) or "附件"
        abs_url = href if href.startswith("http") else f"http://60.29.58.150{href}" if href.startswith("/") else href
        key = f"__LINK__{len(link_protect)}__"
        link_protect[key] = f'<a href="{abs_url}" target="_blank">{txt}</a>'
        return key
    content2 = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content2, flags=re.S | re.I)

    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return f"__TBL__{len(table_protect)-1}__"
    content2 = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content2, flags=re.S | re.I)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0

    content2 = re.sub(r"</p>", "</p>\n\n", content2, flags=re.I)
    content2 = re.sub(r"<br\s*/?>", "\n", content2, flags=re.I)
    content2 = re.sub(r"</?(?:span|font|o:p|st1?:[a-z]+)\b[^>]*>", "", content2, flags=re.I)

    parts = []
    for block in content2.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = plain.replace("\xa0", " ").replace("&nbsp;", " ")
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace(f"__TBL__{i}__", tbl)
        parts.append(b.strip())

    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace(f"__TBL__{i}__", tbl)

    out = re.sub(r"__LINK__\d+__", "", out)
    out = re.sub(r"__TBL__\d+__", "", out)

    out = html_lib.unescape(out)
    out = re.sub(r"[ \t]+\n", "\n", out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    out = re.sub(r"[ \t]{2,}", " ", out)

    seen = set()
    final = []
    for p in out.split("\n\n"):
        key = re.sub(r"\s+", "", re.sub(r"<[^>]+>", "", p))
        if not key:
            continue
        if key in seen:
            continue
        seen.add(key)
        final.append(p)
    out = "\n\n".join(final).strip()

    # append attachments as embedded links
    for url, name in (attachments or []):
        if name:
            out += f'\n\n<p><a href="{url}" target="_blank">{name}</a></p>\n'

    return out, has_table, attachments


def store_record(conn, url, title, content, date, has_table):
    cur = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,))
    if cur.fetchone():
        return False
    try:
        industry = "other"
        try:
            from crawler_lib import classify_industry
            industry = classify_industry(title)
        except Exception:
            pass
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(page_url, source_url, title, content, publish_date, site_name, group_name, script_name, has_table, industry) "
            "VALUES (?,?,?,?,?,?,?,?,?,?)",
            (url, url, title, content, date, SITE_NAME, GROUP_NAME, SCRIPT_NAME, has_table, industry),
        )
        if cur.rowcount > 0:
            summary = re.sub(r"<[^>]+>", "", content)
            summary = html_lib.unescape(summary)
            summary = re.sub(r"\s+", " ", summary).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES (?,?,?,?)",
                (cur.lastrowid, title, SITE_NAME, summary),
            )
            return True
        return False
    except sqlite3.OperationalError as e:
        if "locked" in str(e):
            time.sleep(3)
            return store_record(conn, url, title, content, date, has_table)
        raise


def main():
    import requests
    s = requests.Session()
    s.headers.update(HEADERS)
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=60000")
    conn.execute("PRAGMA journal_mode=WAL")

    new_count = 0
    skip_count = 0
    seen_urls = set()

    for page in range(1, _PAGES + 1):
        try:
            items, total = fetch_list(s, page)
        except Exception as e:
            print(f"Page {page} fetch error: {e}", flush=True)
            break
        print(f"Page {page}: found {len(items)} items (total={total})", flush=True)
        if not items:
            print("Empty page, stop pagination", flush=True)
            break
        for abs_url, title, date, content_html, attachments in items:
            if abs_url in seen_urls:
                continue
            seen_urls.add(abs_url)
            content, has_table, _ = html_to_text(content_html, abs_url, attachments)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            if plain_len < 10:
                print(f"  skip empty: {title[:30]}", flush=True)
                skip_count += 1
                continue
            ok = store_record(conn, abs_url, title, content, date, has_table)
            if ok:
                new_count += 1
                print(f"  + {title[:40]} ({date})", flush=True)
            else:
                skip_count += 1
            time.sleep(0.15)
        time.sleep(0.3)

    conn.commit()
    conn.close()
    print(f"新增: {new_count}", flush=True)
    print(f"跳过: {skip_count}", flush=True)
    print(f"=== {SITE_NAME} done ===", flush=True)


if __name__ == "__main__":
    main()
