#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""安庆市生态环境局 - 建设项目环境影响评价审批 (site/label/8888 JSON API)"""
import sys, os, re, time, json, urllib.parse, sqlite3

BASE_URL = "https://sthjj.anqing.gov.cn"
API_URL = BASE_URL + "/anqing/site/label/8888"
SITE_NAME = "安庆市生态环境局-建设项目环评审批"
GROUP_NAME = "安徽"
SCRIPT_NAME = "crawl_anqing_sthjj.py"
CATEGORY = "建设项目环境影响评价审批"
CAT_IDS = "6999372"

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://sthjj.anqing.gov.cn/public/column/4018218?type=4&catId=6999372&action=list&nav=3",
    "X-Requested-With": "XMLHttpRequest",
}

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

_DAYS_CUTOFF = 365 * 3
_CUTOFF_DATE = time.strftime("%Y-%m-%d", time.localtime(time.time() - _DAYS_CUTOFF * 86400))


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def clean_title(t):
    if not t:
        return ""
    import html as html_lib
    t = html_lib.unescape(t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch_api(page):
    import requests, urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
    params = {
        "labelName": "publicInfoList",
        "siteId": "3902127",
        "organId": "4018218",
        "catIds": CAT_IDS,
        "pageSize": "20",
        "pageIndex": str(page),
        "isDate": "true",
        "dateFormat": "yyyy-MM-dd",
        "length": "50",
        "type": "4",
        "action": "list",
        "result": "",
        "isJson": "true",
        "isSetValue": "true",
    }
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=30)
        if r.status_code != 200:
            print(f"  [API status {r.status_code}]", flush=True)
            return [], 0
        data = r.json()
        items = data.get("data", [])
        total = data.get("total", 0)
        return items, total
    except Exception as e:
        print(f"  [API fail] {e}", flush=True)
        return [], 0


def fetch(url):
    import requests, urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        if r.status_code == 200:
            # Content-Type lacks charset -> requests defaults to ISO-8859-1 -> mojibake
            if "charset" not in (r.headers.get("Content-Type") or "").lower():
                return r.content.decode("utf-8", errors="replace")
            return r.text
    except Exception as e:
        print(f"  [fetch fail] {e}", flush=True)
    return None


def html_table_to_html(table, base_url=""):
    if BeautifulSoup is None:
        return ""
    tbl = BeautifulSoup(str(table), "html.parser")
    for a in tbl.find_all("a"):
        href = a.get("href", "")
        if href and not href.startswith(("http", "javascript", "#")):
            a["href"] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all("img"):
        src = img.get("src", "")
        if src and not src.startswith(("http", "data:", "javascript")):
            img["src"] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_detail(html, url):
    if BeautifulSoup is None:
        return {"title": "", "date": "", "content": "", "attachments": []}
    soup = BeautifulSoup(html, "html.parser")

    title = ""
    t = soup.find("meta", attrs={"name": "ArticleTitle"})
    if t and t.get("content"):
        title = t["content"].strip()
    if not title:
        h = soup.select_one("h1, h2, div.title, div.ArticleTitle")
        if h:
            title = h.get_text(" ", strip=True)
    if not title and soup.title:
        title = soup.title.string.strip()

    date_text = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", md["content"])
        if m:
            date_text = m.group(1)
    if not date_text:
        m = re.search(r"(\d{4}-\d{2}-\d{2})", html)
        if m:
            date_text = m.group(1)

    content_div = soup.select_one("div.details-content, div.content, div.article-content, div.cont, #Zoom, font#Zoom, div.TRS_Editor, div.trs_editor_view, div.text-content, div.article-content-main")
    if not content_div:
        content_div = soup.find("div", class_=re.compile("content|article|details", re.I))
    if not content_div:
        return {"title": title, "date": date_text, "content": "", "attachments": []}

    attachments = []
    parts = []

    for el in content_div.find_all(["p", "pre", "table", "img"], recursive=True):
        if el.name in ("p", "pre") and el.find_parent("table") and el.name == "p":
            continue
        if el.name == "table" and el.find_parent("table"):
            continue

        if el.name in ("p", "pre"):
            text = el.get_text(" ", strip=True)
            if el.name == "pre":
                text = body_text(el)
            for a in el.find_all("a", href=True, appendix=True):
                href = a["href"].strip()
                atitle = a.get_text(strip=True) or a.get("download", "")
                oldsrc = a.get("OLDSRC", "")
                if oldsrc:
                    ahref = (BASE_URL + oldsrc) if oldsrc.startswith("/") else (url.rsplit("/", 1)[0] + "/" + oldsrc)
                else:
                    if href.startswith("./"):
                        ahref = url.rsplit("/", 1)[0] + "/" + href[2:]
                    elif href.startswith("/"):
                        ahref = BASE_URL + href
                    elif href.startswith("http"):
                        ahref = href
                    else:
                        ahref = url.rsplit("/", 1)[0] + "/" + href
                attachments.append({"title": atitle, "url": ahref})
            if text:
                parts.append(f"<p>{text}</p>")

        elif el.name == "table":
            tbl_html = html_table_to_html(el, url)
            if tbl_html:
                parts.append(tbl_html)
        elif el.name == "img" and not el.find_parent("table"):
            src = el.get("src", "")
            alt = el.get("alt", "")
            if src:
                if src.startswith("//"):
                    src = "https:" + src
                elif src.startswith("/"):
                    src = BASE_URL + src
                elif not src.startswith("http"):
                    src = url.rsplit("/", 1)[0] + "/" + src
                parts.append('<a href="%s" target="_blank">%s</a>' % (src, alt or "图片"))

    for a in content_div.find_all("a", href=True):
        href = a["href"].strip()
        if (re.search(r"\.(pdf|docx?|xlsx?|rar|zip)$", href, re.I) or href.startswith("http")) and not a.get("appendix"):
            atitle = a.get_text(strip=True) or a.get("download", "")
            if atitle and href != "./" and not any(att["url"].endswith(href.split("/")[-1]) for att in attachments):
                oldsrc = a.get("OLDSRC", "")
                if oldsrc:
                    ahref = (BASE_URL + oldsrc) if oldsrc.startswith("/") else (url.rsplit("/", 1)[0] + "/" + oldsrc)
                else:
                    if href.startswith("./"):
                        ahref = url.rsplit("/", 1)[0] + "/" + href[2:]
                    elif href.startswith("/"):
                        ahref = BASE_URL + href
                    elif href.startswith("http"):
                        ahref = href
                    else:
                        ahref = url.rsplit("/", 1)[0] + "/" + href
                attachments.append({"title": atitle, "url": ahref})

    # Fallback: bare text
    bare_text = content_div.get_text(" ", strip=True)
    if not parts and bare_text and not content_div.find(["p", "pre", "table", "img"]):
        parts.append(bare_text)

    if not parts:
        link_paras = []
        for a in content_div.find_all("a", href=True):
            href = a["href"].strip()
            if href.startswith("http") and "javascript" not in href:
                atitle = a.get_text(strip=True) or a.get("download", "") or href
                if href.startswith("./"):
                    ahref = url.rsplit("/", 1)[0] + "/" + href[2:]
                elif href.startswith("/"):
                    ahref = BASE_URL + href
                else:
                    ahref = href
                link_paras.append('<p><a href="%s" target="_blank">%s</a></p>' % (ahref, atitle))
        if link_paras:
            parts = link_paras

    seen = set()
    unique_attachments = []
    for att in attachments:
        if att["url"] not in seen:
            seen.add(att["url"])
            unique_attachments.append(att)

    content = "\n\n".join(parts)
    return {"title": title, "date": date_text, "content": content, "attachments": unique_attachments}


def push_to_searchdb(items):
    if JSONL_PATH:
        with open(JSONL_PATH, "a", encoding="utf-8") as f:
            for it in items:
                f.write(json.dumps(it, ensure_ascii=False) + "\n")
        print("  [jsonl] wrote %d" % len(items))
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    new_count = 0
    for it in items:
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (it["url"],))
        if c.fetchone():
            continue
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank, script_name) VALUES (?,?,?,?,?,?,?,?,?,?)",
            (it["title"], SITE_NAME, GROUP_NAME, it["url"], it["date"], it["content"], "", it["attachments_json"], it["date_rank"], SCRIPT_NAME),
        )
        if c.rowcount:
            new_count += 1
    conn.commit()
    conn.close()
    print("新增: %d" % new_count)


def main():
    print("=== %s done 前置 ===" % SITE_NAME)
    all_items = []
    total_records = 0
    for page in range(1, _PAGES + 1):
        items, total = fetch_api(page)
        if not items:
            print(f"  page {page} empty/failed")
            continue
        total_records = total
        for it in items:
            title = clean_title(it.get("title", ""))
            link = it.get("link", "")
            date = (it.get("publishDate") or it.get("createDate") or "")[:10]
            if not title or not link:
                continue
            if not link.startswith("http"):
                link = BASE_URL + link
            all_items.append((link, title, date))
        print(f"  page {page}: {len(items)} items (total={total_records})", flush=True)

    filtered = []
    for href, title, date in all_items:
        if date and date < _CUTOFF_DATE:
            continue
        filtered.append((href, title, date))
    print("列表总数: %d, 3年截断后: %d" % (len(all_items), len(filtered)))

    results = []
    for href, title, date in filtered:
        html = fetch(href)
        if not html:
            results.append({"url": href, "title": title, "date": date, "content": "", "attachments": "", "attachments_json": "[]", "date_rank": 0})
            continue
        d = parse_detail(html, href)
        atts = d.get("attachments", [])
        att_text = "; ".join(a["url"] for a in atts)
        date_final = d.get("date") or date
        rank = 0
        m = re.search(r"(\d{4}-\d{2}-\d{2})", date_final)
        if m:
            rank = int(m.group(1).replace("-", ""))
        results.append({
            "url": href, "title": d.get("title") or title, "date": date_final,
            "content": d.get("content", ""), "attachments": att_text,
            "attachments_json": json.dumps(atts, ensure_ascii=False),
            "date_rank": rank,
        })

    empty = sum(1 for r in results if not r["content"])
    att_count = sum(1 for r in results if r["attachments"])
    print("[RESULT] %s: %d new, %d empty content, %d attachments" % (SITE_NAME, len(results), empty, att_count))
    push_to_searchdb(results)


if __name__ == "__main__":
    main()
