#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""淄博市生态环境局 - 受理公示 (dataproxy.jsp XML)"""
import sys, os, re, time, json, urllib.parse, sqlite3

BASE_URL = "https://epb.zibo.gov.cn"
API_URL = BASE_URL + "/module/web/jpage/dataproxy.jsp"
SITE_NAME = "淄博市生态环境局-受理公示"
GROUP_NAME = "山东"
SCRIPT_NAME = "crawl_epb_zibo.py"
CATEGORY = "受理公示"

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")

BASE_PARAMS = "appid=1&webid=11&path=/&webname=%E6%B7%84%E5%8D%9A%E5%B8%82%E7%94%9F%E6%80%81%E7%8E%AF%E5%A2%83%E5%B1%80&permissiontype=0&columnid=2210&unitid=61151"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

_DAYS_CUTOFF = 365 * 3
_CUTOFF_DATE = time.strftime("%Y-%m-%d", time.localtime(time.time() - _DAYS_CUTOFF * 86400))


def clean_title(t):
    if not t:
        return ""
    import html as html_lib
    t = html_lib.unescape(t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch_api(page):
    import requests, urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
    url = "%s?page=%d&perPage=50&%s" % (API_URL, page, BASE_PARAMS)
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        if r.status_code != 200:
            return None, 0
        import xml.etree.ElementTree as ET
        tree = ET.fromstring(r.content)
        total = int(tree.findtext("totalrecord", "0") or 0)
        records = tree.findall(".//record")
        return records, total
    except Exception as e:
        print("  [API fail] %s" % e, flush=True)
        return None, 0


def parse_records(records):
    items = []
    for rec in records:
        cdata = rec.text or ""
        m_url = re.search(r"href='([^']+)'", cdata)
        m_title = re.search(r"title='([^']+)'", cdata)
        m_date = re.search(r">(\d{4}-\d{2}-\d{2})<", cdata)
        if not m_url or not m_title:
            continue
        href = m_url.group(1)
        title = clean_title(m_title.group(1))
        date = m_date.group(1) if m_date else ""
        if not title or len(title) < 4:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append((href, title, date))
    return items


def fetch(url):
    import requests, urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        if r.status_code == 200:
            return r.text
    except Exception as e:
        print("  [fetch fail] %s" % e, flush=True)
    return None


def html_table_to_html(table, base_url=""):
    if BeautifulSoup is None:
        return ""
    tbl = BeautifulSoup(str(table), "html.parser")
    for a in tbl.find_all("a"):
        href = a.get("href", "")
        if href and not href.startswith(("http", "javascript", "#")):
            a["href"] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all("img"):
        src = img.get("src", "")
        if src and not src.startswith(("http", "data:", "javascript")):
            img["src"] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_detail(html, url):
    if BeautifulSoup is None:
        return {"title": "", "date": "", "content": "", "attachments": []}
    soup = BeautifulSoup(html, "html.parser")

    title = ""
    t = soup.find("meta", attrs={"name": "ArticleTitle"})
    if t and t.get("content"):
        title = t["content"].strip()
    if not title:
        h = soup.select_one("h1, h2, div.title, div.ArticleTitle")
        if h:
            title = h.get_text(" ", strip=True)
    if not title and soup.title:
        title = soup.title.string.strip()

    date_text = ""
    md = soup.find("meta", attrs={"name": "PubDate"})
    if md and md.get("content"):
        m = re.search(r"(\d{4}-\d{2}-\d{2})", md["content"])
        if m:
            date_text = m.group(1)
    if not date_text:
        m = re.search(r"(\d{4}-\d{2}-\d{2})", html)
        if m:
            date_text = m.group(1)

    content_div = soup.select_one("div.details-content, div.content, div.article-content, div.cont, #Zoom, font#Zoom, div.TRS_Editor, div.trs_editor_view")
    if not content_div:
        content_div = soup.find("div", class_=re.compile("content|article|details", re.I))
    if not content_div:
        return {"title": title, "date": date_text, "content": "", "attachments": []}

    attachments = []
    parts = []

    for el in content_div.find_all(["p", "pre", "table", "img"], recursive=True):
        if el.name in ("p", "pre") and el.find_parent("table") and el.name == "p":
            continue
        if el.name == "table" and el.find_parent("table"):
            continue

        if el.name in ("p", "pre"):
            text = el.get_text(" ", strip=True)
            if el.name == "pre":
                text = el.get_text("\n", strip=True)
            for a in el.find_all("a", href=True, appendix=True):
                href = a["href"].strip()
                atitle = a.get_text(strip=True) or a.get("download", "")
                oldsrc = a.get("OLDSRC", "")
                if oldsrc:
                    ahref = (BASE_URL + oldsrc) if oldsrc.startswith("/") else (url.rsplit("/", 1)[0] + "/" + oldsrc)
                else:
                    if href.startswith("./"):
                        ahref = url.rsplit("/", 1)[0] + "/" + href[2:]
                    elif href.startswith("/"):
                        ahref = BASE_URL + href
                    elif href.startswith("http"):
                        ahref = href
                    else:
                        ahref = url.rsplit("/", 1)[0] + "/" + href
                attachments.append({"title": atitle, "url": ahref})
            if text:
                parts.append(text)

        elif el.name == "table":
            tbl_html = html_table_to_html(el, url)
            if tbl_html:
                parts.append(tbl_html)
        elif el.name == "img":
            src = el.get("src", "")
            alt = el.get("alt", "")
            if src:
                if src.startswith("//"):
                    src = "https:" + src
                elif src.startswith("/"):
                    src = BASE_URL + src
                elif not src.startswith("http"):
                    src = url.rsplit("/", 1)[0] + "/" + src
                parts.append('<a href="%s" target="_blank">%s</a>' % (src, alt or "图片"))

    for a in content_div.find_all("a", href=True):
        href = a["href"].strip()
        if (re.search(r"\.(pdf|docx?|xlsx?|rar|zip)$", href, re.I) or href.startswith("http")) and not a.get("appendix"):
            atitle = a.get_text(strip=True) or a.get("download", "")
            if atitle and href != "./" and not any(att["url"].endswith(href.split("/")[-1]) for att in attachments):
                oldsrc = a.get("OLDSRC", "")
                if oldsrc:
                    ahref = (BASE_URL + oldsrc) if oldsrc.startswith("/") else (url.rsplit("/", 1)[0] + "/" + oldsrc)
                else:
                    if href.startswith("./"):
                        ahref = url.rsplit("/", 1)[0] + "/" + href[2:]
                    elif href.startswith("/"):
                        ahref = BASE_URL + href
                    elif href.startswith("http"):
                        ahref = href
                    else:
                        ahref = url.rsplit("/", 1)[0] + "/" + href
                attachments.append({"title": atitle, "url": ahref})

    # Fallback: bare text directly under container
    bare_text = content_div.get_text(" ", strip=True)
    if not parts and bare_text and not content_div.find(["p", "pre", "table", "img"]):
        for a in content_div.find_all("a", href=True, appendix=True):
            href = a["href"].strip()
            atitle = a.get_text(strip=True) or a.get("download", "")
            oldsrc = a.get("OLDSRC", "")
            if oldsrc:
                ahref = (BASE_URL + oldsrc) if oldsrc.startswith("/") else (url.rsplit("/", 1)[0] + "/" + oldsrc)
            else:
                if href.startswith("./"):
                    ahref = url.rsplit("/", 1)[0] + "/" + href[2:]
                elif href.startswith("/"):
                    ahref = BASE_URL + href
                elif href.startswith("http"):
                    ahref = href
                else:
                    ahref = url.rsplit("/", 1)[0] + "/" + href
            attachments.append({"title": atitle, "url": ahref})
        parts.append(bare_text)

    if not parts:
        link_paras = []
        for a in content_div.find_all("a", href=True):
            href = a["href"].strip()
            if href.startswith("http") and "javascript" not in href:
                atitle = a.get_text(strip=True) or a.get("download", "") or href
                if href.startswith("./"):
                    ahref = url.rsplit("/", 1)[0] + "/" + href[2:]
                elif href.startswith("/"):
                    ahref = BASE_URL + href
                else:
                    ahref = href
                link_paras.append('<p><a href="%s" target="_blank">%s</a></p>' % (ahref, atitle))
        if link_paras:
            parts = link_paras

    seen = set()
    unique_attachments = []
    for att in attachments:
        if att["url"] not in seen:
            seen.add(att["url"])
            unique_attachments.append(att)

    content = "\n\n".join(parts)
    return {"title": title, "date": date_text, "content": content, "attachments": unique_attachments}


def push_to_searchdb(items):
    if JSONL_PATH:
        with open(JSONL_PATH, "a", encoding="utf-8") as f:
            for it in items:
                f.write(json.dumps(it, ensure_ascii=False) + "\n")
        print("  [jsonl] wrote %d" % len(items))
        return

    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    new_count = 0
    for it in items:
        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (it["url"],))
        if c.fetchone():
            continue
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (title, site_name, group_name, page_url, publish_date, content, summary, attachments, date_rank, script_name) VALUES (?,?,?,?,?,?,?,?,?,?)",
            (it["title"], SITE_NAME, GROUP_NAME, it["url"], it["date"], it["content"], "", it["attachments_json"], it["date_rank"], SCRIPT_NAME),
        )
        if c.rowcount:
            new_count += 1
    conn.commit()
    conn.close()
    print("新增: %d" % new_count)


def main():
    print("=== %s done 前置 ===" % SITE_NAME)
    all_items = []
    total_records = 0
    for page in range(1, _PAGES + 1):
        records, total = fetch_api(page)
        if records is None:
            print("  page %d failed" % page)
            continue
        total_records = total
        items = parse_records(records)
        all_items.extend(items)
        print("  page %d: %d items" % (page, len(items)), flush=True)

    filtered = []
    for href, title, date in all_items:
        if date and date < _CUTOFF_DATE:
            continue
        filtered.append((href, title, date))
    print("列表总数: %d, 3年截断后: %d" % (len(all_items), len(filtered)))

    results = []
    for href, title, date in filtered:
        html = fetch(href)
        if not html:
            results.append({"url": href, "title": title, "date": date, "content": "", "attachments": "", "attachments_json": "[]", "date_rank": 0})
            continue
        d = parse_detail(html, href)
        atts = d.get("attachments", [])
        att_text = "; ".join(a["url"] for a in atts)
        date_final = d.get("date") or date
        rank = 0
        m = re.search(r"(\d{4}-\d{2}-\d{2})", date_final)
        if m:
            rank = int(m.group(1).replace("-", ""))
        results.append({
            "url": href, "title": d.get("title") or title, "date": date_final,
            "content": d.get("content", ""), "attachments": att_text,
            "attachments_json": json.dumps(atts, ensure_ascii=False),
            "date_rank": rank,
        })

    empty = sum(1 for r in results if not r["content"])
    att_count = sum(1 for r in results if r["attachments"])
    print("[RESULT] %s: %d new, %d empty content, %d attachments" % (SITE_NAME, len(results), empty, att_count))
    push_to_searchdb(results)


if __name__ == "__main__":
    main()
