#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""淄博市发展和改革委员会 - 公示公告 (dataproxy.jsp XML)"""
import sys, os, re, time, json, urllib.parse, sqlite3

BASE_URL = "http://fgw.zibo.gov.cn"
API_URL = BASE_URL + "/module/web/jpage/dataproxy.jsp"
SITE_NAME = "淄博市发改委-公示公告"
GROUP_NAME = "山东"
SCRIPT_NAME = "crawl_fgw_zibo.py"
CATEGORY = "公示公告"

try:
    from bs4 import BeautifulSoup
except ImportError:
    BeautifulSoup = None

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
JSONL_PATH = os.environ.get("JSONL_PATH", "")

BASE_PARAMS = "appid=1&webid=43&path=/&webname=%E6%B7%84%E5%8D%9A%E5%B8%82%E5%8F%91%E5%B1%95%E5%92%8C%E6%94%B9%E9%9D%A9%E5%A7%94%E5%91%98%E4%BC%9A&permissiontype=0&columnid=951&unitid=57113"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

_MAX_PAGES = 1
for i, a in enumerate(sys.argv):
    if a.startswith("--pages="):
        try:
            _MAX_PAGES = int(a.split("=", 1)[1])
        except ValueError:
            pass
    elif a == "--pages" and i + 1 < len(sys.argv):
        try:
            _MAX_PAGES = int(sys.argv[i + 1])
        except ValueError:
            pass
_PAGES = _MAX_PAGES if _MAX_PAGES >= 1 else 1

_DAYS_CUTOFF = 365 * 3
_CUTOFF_DATE = time.strftime("%Y-%m-%d", time.localtime(time.time() - _DAYS_CUTOFF * 86400))


def clean_title(t):
    if not t:
        return ""
    import html as html_lib
    t = html_lib.unescape(t)
    t = re.sub(r"\s+", " ", t)
    return t.strip()


def fetch_api(page):
    import requests, urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
    url = "%s?page=%d&perPage=50&%s" % (API_URL, page, BASE_PARAMS)
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        if r.status_code != 200:
            return None, 0
        import xml.etree.ElementTree as ET
        tree = ET.fromstring(r.content)
        total = int(tree.findtext("totalrecord", "0") or 0)
        records = tree.findall(".//record")
        return records, total
    except Exception as e:
        print("  [API fail] %s" % e, flush=True)
        return None, 0


def parse_records(records):
    items = []
    for rec in records:
        cdata = rec.text or ""
        m_url = re.search(r"href='([^']+)'", cdata)
        m_title = re.search(r"title='([^']+)'", cdata)
        m_date = re.search(r">(\d{4}-\d{2}-\d{2})<", cdata)
        if not m_url or not m_title:
            continue
        href = m_url.group(1)
        title = clean_title(m_title.group(1))
        date = m_date.group(1) if m_date else ""
        if not title or len(title) < 4:
            continue
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append((href, title, date))
    return items


def fetch(url):
    import requests, urllib3
    urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        if r.status_code == 200:
            r.encoding = "utf-8"
            return r.text
    except Exception as e:
        print("  [fetch fail] %s" % e, flush=True)
    return None


def html_to_text(content, page_url):
    if not content:
        return "", 0, []
    import html as html_lib
    content = re.sub(r"<!--.*?-->", "", content, flags=re.S)
    content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.S | re.I)
    content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.S | re.I)
    attachments = []
    link_protect = {}
    if BeautifulSoup:
        try:
            _soup = BeautifulSoup(content, "html.parser")
            for a in _soup.find_all("a", href=True):
                if a.get("appendix") or re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps)", a.get("href", ""), re.I):
                    href = a["href"]
                    real_href = a.get("oldsrc") or href
                    txt = a.get_text(strip=True) or os.path.basename(real_href.split("?")[0])
                    abs_url = urllib.parse.urljoin(page_url, real_href)
                    attachments.append((abs_url, txt))
                    key = "__ATTACH__%d__" % len(link_protect)
                    link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
                    a.replace_with(key)
            content = str(_soup)
        except Exception:
            pass

    def _link_repl(m):
        href = m.group(1)
        if href.startswith("javascript:"):
            return ""
        inner = m.group(2)
        txt = re.sub(r"<[^>]+>", "", inner)
        txt = html_lib.unescape(txt).strip()
        if re.search(r"\.(pdf|docx?|xlsx?|zip|rar|wps)", href, re.I) or "download" in href.lower() or "attachment" in href.lower():
            if not txt:
                txt = os.path.basename(href.split("?")[0]) or "附件"
            abs_url = urllib.parse.urljoin(page_url, href)
            attachments.append((abs_url, txt))
            key = "__LINK__%d__" % len(link_protect)
            link_protect[key] = '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
            return key
        abs_url = urllib.parse.urljoin(page_url, href)
        return '<a href="%s" target="_blank">%s</a>' % (abs_url, txt)
    content = re.sub(r'<a\s[^>]*href="([^"]+)"[^>]*>(.*?)</a>', _link_repl, content, flags=re.S | re.I)

    table_protect = []
    def _tbl_repl(m):
        table_protect.append(m.group(0))
        return "__TBL__%d__" % (len(table_protect) - 1)
    has_table = 1 if re.search(r"<table[^>]*>", content, re.I) else 0
    content = re.sub(r"<table[^>]*>.*?</table>", _tbl_repl, content, flags=re.S | re.I)

    content = re.sub(r"</p>", "</p>\n\n", content, flags=re.I)
    content = re.sub(r"<br\s*/?>", "\n", content, flags=re.I)
    content = re.sub(r"</?(?:span|font|o:p)\b[^>]*>", "", content, flags=re.I)

    parts = []
    for block in content.split("\n\n"):
        b = block.strip()
        if not b:
            continue
        plain = re.sub(r"<[^>]+>", "", b)
        plain = html_lib.unescape(plain)
        plain = re.sub(r"[\s\u200b\u200c\u200d\ufeff]+", "", plain)
        if not plain:
            continue
        for k, v in link_protect.items():
            b = b.replace(k, v)
        for i, tbl in enumerate(table_protect):
            b = b.replace("__TBL__%d__" % i, tbl)
        parts.append(b.strip())
    out = "\n\n".join(parts)
    for k, v in link_protect.items():
        out = out.replace(k, v)
    for i, tbl in enumerate(table_protect):
        out = out.replace("__TBL__%d__" % i, tbl)
    out = re.sub(r"__LINK__\d+__|__TBL__\d+__|__ATTACH__\d+__", "", out)
    out = html_lib.unescape(out)
    out = re.sub(r"\n{3,}", "\n\n", out)
    return out.strip(), has_table, attachments


def parse_detail(html_text, page_url):
    title = ""; date = ""; content_html = ""
    if BeautifulSoup:
        soup = BeautifulSoup(html_text, "html.parser")
        d = soup.select_one("div.TRS_Editor, div.content, div.article-content, #zoom, div.maintext, div.con_text")
        if d:
            content_html = "".join(str(c) for c in d.contents)
        t = soup.find("meta", attrs={"name": re.compile("ArticleTitle", re.I)})
        if t and t.get("content"):
            title = clean_title(t["content"])
        md = soup.find("meta", attrs={"name": re.compile("PubDate", re.I)})
        if md and md.get("content"):
            dm = re.search(r"([0-9]{4}-[0-9]{2}-[0-9]{2})", md["content"])
            if dm:
                date = dm.group(1)
        if not title:
            h1 = soup.find("h1") or soup.find("h2")
            if h1:
                title = clean_title(h1.get_text())
    return title, date, content_html


def main():
    sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
    from crawler_lib import push_to_searchdb
    new_count = 0
    skip_count = 0
    seen_urls = set()
    batch = []
    jsonl_rows = []

    for page in range(1, _PAGES + 1):
        records, total = fetch_api(page)
        if records is None:
            print("Page %d API failed" % page, flush=True)
            break
        items = parse_records(records)
        if not items:
            print("Page %d: empty, stop" % page, flush=True)
            break
        print("Page %d: %d items (total=%s)" % (page, len(items), total), flush=True)
        for url, title, date in items:
            if url in seen_urls:
                continue
            seen_urls.add(url)
            if date and date < _CUTOFF_DATE:
                continue
            dhtml = fetch(url)
            if not dhtml:
                skip_count += 1
                continue
            dtitle, ddate, content_html = parse_detail(dhtml, url)
            if not dtitle:
                dtitle = title
            if not ddate:
                ddate = date
            if ddate and ddate < _CUTOFF_DATE:
                skip_count += 1
                continue
            content, has_table, atts = html_to_text(content_html, url)
            plain_len = len(re.sub(r"<[^>]+>", "", content).strip())
            if plain_len < 10:
                print("  skip empty: %s" % dtitle[:40], flush=True)
                skip_count += 1
                continue
            item = {
                "site_name": SITE_NAME, "source_url": url, "url": url,
                "title": dtitle, "pub_date": ddate, "summary": "",
                "content": content, "category": CATEGORY, "tags": "",
                "group_name": GROUP_NAME,
                "attachments": "; ".join("%s|%s" % (u, t) for u, t in atts) if atts else "",
            }
            if JSONL_PATH:
                jsonl_rows.append(item)
                print("  JSONL + %s (%s)" % (dtitle[:40], ddate), flush=True)
            else:
                batch.append(item)
                print("  + %s (%s)" % (dtitle[:40], ddate), flush=True)
            new_count += 1
            time.sleep(0.05)

    if JSONL_PATH:
        with open(JSONL_PATH, "w", encoding="utf-8") as f:
            for row in jsonl_rows:
                f.write(json.dumps(row, ensure_ascii=False) + "\n")
        print("JSONL written: %s (%d rows)" % (JSONL_PATH, len(jsonl_rows)), flush=True)
    else:
        push_to_searchdb(batch, batch_label=SCRIPT_NAME)

    print("新增: %d" % new_count, flush=True)
    print("跳过: %d" % skip_count, flush=True)
    print("=== %s done ===" % SITE_NAME, flush=True)


if __name__ == "__main__":
    main()
