#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
金昌市生态环境局 - 已批准项目公示
URL: https://sthj.jcs.gov.cn/xmsp/ypzxmgs/index.html
CMS: JPAAS publish API
列表: GET /api-gateway/jpaas-publish-server/front/page/build/unit + paramJson
"""
import sys, os, re, time, json, urllib.parse, sqlite3

BASE_URL = "https://sthj.jcs.gov.cn"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
SITE_NAME = "金昌市生态环境局-已批准项目公示"
GROUP_NAME = "甘肃"
SCRIPT_NAME = "crawl_jcs_ypzxmgs.py"
CATEGORY = "环评审批"

WEB_ID = "47f74ec6fa054832961ddb79491f6065"
PAGE_ID = "aa4be0e6b446412c9a0cbbece91f53a9"
TPL_SET = "a7ce57f894724fb1970fb836e7b631ff"
TAG_ID = "分页"
LIST_PAGE = "/xmsp/ypzxmgs/index.html"

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Referer": BASE_URL + LIST_PAGE,
}

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

_DAYS_CUTOFF = 365 * 3
_CUTOFF_DATE = time.strftime("%Y-%m-%d", time.localtime(time.time() - _DAYS_CUTOFF * 86400))


def clean_title(t):
    if not t:
        return ""
    import html as html_lib
    t = html_lib.unescape(t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch_api(page):
    import requests, urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
    params = {
        "parseType": "bulidstatic",
        "webId": WEB_ID,
        "tplSetId": TPL_SET,
        "pageType": "column",
        "tagId": TAG_ID,
        "editType": "null",
        "pageId": PAGE_ID,
    }
    params["paramJson"] = json.dumps({"pageNo": page, "pageSize": 20}, ensure_ascii=False)
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30, verify=False)
        if r.status_code != 200:
            return None, 0
        data = r.json()
        if not data.get("success"):
            return None, 0
        html = data["data"]["html"]
        m = re.search(r'count="(\d+)"', html)
        total = int(m.group(1)) if m else 0
        return html, total
    except Exception as e:
        print("  [API fail] %s" % e, flush=True)
        return None, 0


def parse_list(html):
    items = []
    if not BeautifulSoup:
        return items
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"].strip()
        title = a.get("title", "") or a.get_text(strip=True)
        date = ""
        span = li.find("span")
        if span:
            date = span.get_text(strip=True)
        title = clean_title(title)
        if not title or len(title) < 4:
            continue
        if not href.startswith("http"):
            href = urllib.parse.urljoin(BASE_URL, href)
        items.append((href, title, date))
    return items


def fetch(url):
    import requests, urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        if r.status_code == 200:
            r.encoding = "utf-8"
            return r.text
    except Exception as e:
        print("  [fetch fail] %s" % e, flush=True)
    return None


def html_to_text(content, page_url):
    if not content:
        return "", 0, []
    import html as html_lib
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)
    attachments = []
    link_protect = {}
    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for a in _soup.find_all("a", href=True):
                if a.get("appendix") or re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps)|fileName=.*\.(pdf|doc)", a.get("href", ""), re.I):
                    href = a["href"]
                    real_href = a.get("oldsrc") or href
                    txt = a.get_text(strip=True) or os.path.basename(real_href.split("?")[0])
                    abs_url = urllib.parse.urljoin(page_url, real_href)
                    attachments.append((abs_url, txt))
                    key = "__ATTACH__%d__" % len(link_protect)
                    link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
                    a.replace_with(key)
            content = str(_soup)
        except Exception:
            pass

    def _link_repl(m):
        href = m.group(1)
        if href.startswith("javascript:"):
            return ""
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        if re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps)|fileName=.*\.(pdf|doc)", href, re.I) or "download" in href.lower() or "attachment" in href.lower():
            if not txt:
                txt = os.path.basename(href.split("?")[0]) or "附件"
            abs_url = urllib.parse.urljoin(page_url, href)
            attachments.append((abs_url, txt))
            key = "__LINK__%d__" % len(link_protect)
            link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
            return key
        abs_url = urllib.parse.urljoin(page_url, href)
        return '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
    content = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content, flags=re.S | re.I)

    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return "__TBL__%d__" % (len(table_protect) - 1)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0
    content = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content, flags=re.S | re.I)

    content = re.sub(r"</p>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"<br\s*/?>", "\n", content, flags=re.I)
    content = re.sub(r"</?(?:span|font|o:p)\b[^>]*>", "", content, flags=re.I)

    parts = []
    for block in content.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace("__TBL__%d__" % i, tbl)
        parts.append(b.strip())
    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace("__TBL__%d__" % i, tbl)
    out = re.sub(r"__LINK__\d+__|__TBL__\d+__|__ATTACH__\d+__", "", out)
    out = html_lib.unescape(out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    return out.strip(), has_table, attachments


def parse_detail(html_text, page_url):
    title = ""; date = ""; content_html = ""
    if BeautifulSoup:
        soup = BeautifulSoup(html_text, "html.parser")
        d = soup.select_one("div.ty-content, div.article-content, div.content")
        if d:
            content_html = "".join(str(c) for c in d.contents)
        t = soup.find("meta", attrs={"name": re.compile("ArticleTitle", re.I)})
        if t and t.get("content"):
            title = clean_title(t["content"])
        md = soup.find("meta", attrs={"name": re.compile("PubDate", re.I)})
        if md and md.get("content"):
            dm = re.search(r"([0-9]{4}-[0-9]{2}-[0-9]{2})", md["content"])
            if dm:
                date = dm.group(1)
        if not title:
            h1 = soup.find("h1") or soup.find("h2")
            if h1:
                title = clean_title(h1.get_text())
    return title, date, content_html


def main():
    sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
    from crawler_lib import push_to_searchdb
    new_count = 0
    skip_count = 0
    seen_urls = set()
    batch = []
    jsonl_rows = []

    for page in range(1, _PAGES + 1):
        html, total = fetch_api(page)
        if html is None:
            print("Page %d API failed" % page, flush=True)
            break
        items = parse_list(html)
        if not items:
            print("Page %d: empty, stop" % page, flush=True)
            break
        print("Page %d: %d items (total=%s)" % (page, len(items), total), flush=True)
        for url, title, date in items:
            if url in seen_urls:
                continue
            seen_urls.add(url)
            if date and date < _CUTOFF_DATE:
                continue
            dhtml = fetch(url)
            if not dhtml:
                skip_count += 1
                continue
            dtitle, ddate, content_html = parse_detail(dhtml, url)
            if not dtitle:
                dtitle = title
            if not ddate:
                ddate = date
            if ddate and ddate < _CUTOFF_DATE:
                skip_count += 1
                continue
            content, has_table, atts = html_to_text(content_html, url)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            if plain_len < 10:
                print("  skip empty: %s" % dtitle[:40], flush=True)
                skip_count += 1
                continue
            item = {
                "site_name": SITE_NAME, "source_url": url, "url": url,
                "title": dtitle, "pub_date": ddate, "summary": "",
                "content": content, "category": CATEGORY, "tags": "",
                "group_name": GROUP_NAME,
                "attachments": "; ".join("%s|%s" % (u, t) for u, t in atts) if atts else "",
            }
            if JSONL_PATH:
                jsonl_rows.append(item)
                print("  JSONL + %s (%s)" % (dtitle[:40], ddate), flush=True)
            else:
                batch.append(item)
                print("  + %s (%s)" % (dtitle[:40], ddate), flush=True)
            new_count += 1
            time.sleep(0.05)

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print("JSONL written: %s (%d rows)" % (JSONL_PATH, len(jsonl_rows)), flush=True)
    else:
        push_to_searchdb(batch, batch_label=SCRIPT_NAME)

    print("新增: %d" % new_count, flush=True)
    print("跳过: %d" % skip_count, flush=True)
    print("=== %s done ===" % SITE_NAME, flush=True)


if __name__ == "__main__":
    main()
