#!/usr/bin/env python3
"""


许昌市人民政府 - 环境保护
https://www.xuchang.gov.cn/govxxgk/11411000005747138B/openSubPage.html?specialurl=...
EpointWebBuilder 系统，分页列表 + API详情
"""
import requests
import re
import sys
import os
import json
import time
from datetime import datetime, timedelta

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
CAT_PATH = "/govxxgk/11411000005747138B/category/001/001002/001002019/001002019001"
LIST_URL = "https://www.xuchang.gov.cn" + CAT_PATH + "/govlist.html"
API_URL = "https://www.xuchang.gov.cn/EpointWebBuilder/zNJSAction.action?cmd=getOpenDetail"
SITE_GUID = "7eb5f7f1-9041-43ad-8e13-8fcb82ea831a"
SITE_NAME = "xuchang_hjbh"

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
})

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
BS = chr(92)  # backslash - HTML uses \' for escaped single quotes


def get_total_pages(html):
    m = re.search(r'<a[^>]*href="[^"]*/(\d+)\.html"[^>]*>尾页</a>', html)
    if m:
        return int(m.group(1))
    pages = re.findall(r'<li class="wb-page-li wb-page-item "><a href="[^"]*/(\d+)\.html">', html)
    return max(int(p) for p in pages) if pages else 1


def parse_list_page(html):
    """解析列表页：提取UUID、标题、日期"""
    bs = chr(92)
    items = []

    # UUIDs from linkToNew(\'UUID\',\'DEPT\') 调用
    uuid_pat = r"linkToNew\(" + bs + "'([^" + bs + "']+)" + bs + "'"
    uuids = re.findall(uuid_pat, html)
    print(f"  UUIDs found: {len(uuids)}")

    # 标题从 <a class="ewb-infoname" ...> 提取
    titles = re.findall(r'<a class="ewb-infoname"[^>]*>(.*?)</a>', html, re.DOTALL)
    titles = [re.sub(r"<[^>]+>", "", t).strip() for t in titles]
    titles = [re.sub(r"\s+", " ", t) for t in titles]

    # 日期
    dates = re.findall(r'<td class="ewb-rqnr">\s*<span>([^<]+)</span>\s*</td>', html)

    for i, uuid in enumerate(uuids):
        item = {"uuid": uuid}
        if i < len(titles):
            item["title"] = titles[i]
        else:
            item["title"] = "untitled"
        if i < len(dates):
            item["date"] = dates[i].strip()
        items.append(item)

    return items


def fetch_detail(uuid):
    try:
        r = session.post(API_URL, data={"infoid": uuid, "siteguid": SITE_GUID}, timeout=30)
        r.encoding = "utf-8"
        data = r.json()
        custom = json.loads(data.get("custom", "{}"))
        if custom and custom.get("data"):
            return custom["data"][0]
    except Exception as e:
        print("  API fail " + uuid[:20] + ": " + str(e))
    return None


def insert_item(item, detail, conn):
    c = conn.cursor()
    uuid = item["uuid"]
    title = detail.get("title", item["title"]) if detail else item["title"]
    pub_date = detail.get("infodate", item.get("date", "")) if detail else item.get("date", "")
    content = detail.get("infocontent", "") if detail else ""
    attachments = detail.get("attach", []) if detail else []

    attach_list = []
    for a in attachments:
        if isinstance(a, dict):
            name = a.get("name", "")
            guid = a.get("attachGuid", "")
            attach_list.append({
                "url": "http://admin.xuchang.gov.cn:8080/EpointWebBuilder/frame/base/attach/attachdown.jspx?attachGuid=" + guid,
                "title": name
            })

    attach_json = json.dumps(attach_list, ensure_ascii=False) if attach_list else ""

    # If infocontent is empty but has attachments, use attachment links as content
    text_content = re.sub(r"<[^>]+>", "", content).strip() if content else ""
    if (not content or len(text_content) < 50) and attach_list:
        attach_html = "<br>\n".join(
            f'<a href="{a["url"]}" target="_blank">{a["title"]}</a>'
            for a in attach_list
        )
        content = f"<h3>{title}</h3>\n<p>附件：{attach_html}</p>"
        text_content = title

    if not content or len(text_content) < 50:
        print("  Short content: " + title[:30] + " (" + str(len(text_content)) + " chars)")

    source_url = "https://www.xuchang.gov.cn/openDetailDynamic.html?infoid=" + uuid
    try:
        c.execute("""
            INSERT OR IGNORE INTO gov_raw
            (title, page_url, source_url, content, publish_date, site_name)
            VALUES (?, ?, ?, ?, ?, ?)
        """, (title, source_url, source_url, content, pub_date, SITE_NAME))
        affected = c.rowcount
        conn.commit()
        if affected > 0:
            print("  OK " + title[:30] + "...")
        return affected
    except Exception as e:
        print("  Insert fail: " + str(e))
        return 0


def crawl(days_back=365):
    import sqlite3
    now = datetime.now()
    cutoff = now - timedelta(days=days_back)

    print("Fetching list...")
    r = session.get(LIST_URL, timeout=30)
    r.encoding = "utf-8"
    total_pages = get_total_pages(r.text)
    print("Total pages: " + str(total_pages))

    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    total = skip_count = error_count = 0

    for page in range(min(total_pages, _MAX_PG or total_pages)):
        if page == 0:
            html = r.text
        else:
            url = "https://www.xuchang.gov.cn" + CAT_PATH + "/" + str(page) + ".html"
            try:
                r2 = session.get(url, timeout=30)
                r2.encoding = "utf-8"
                html = r2.text
            except Exception as e:
                print("  Page " + str(page+1) + " fail: " + str(e))
                error_count += 1
                continue

        items = parse_list_page(html)
        print("  Page " + str(page+1) + "/" + str(total_pages) + " (" + str(len(items)) + " items)")

        for item in items:
            ds = item.get("date", "")
            if ds:
                try:
                    dt = datetime.strptime(ds, "%Y-%m-%d")
                    if dt < cutoff and days_back < 3650:
                        skip_count += 1
                        continue
                except ValueError:
                    pass

            detail = fetch_detail(item["uuid"])
            if not detail:
                error_count += 1
                detail = {"title": item["title"], "infocontent": "", "infodate": item.get("date", ""), "attach": []}

            affected = insert_item(item, detail, conn)
            if affected > 0:
                total += 1
            time.sleep(0.3)

    conn.close()
    print("Done: new=" + str(total) + " skip=" + str(skip_count) + " err=" + str(error_count))
    return total


if __name__ == "__main__":
    days = int(sys.argv[1]) if len(sys.argv) > 1 else 365
    crawl(days_back=days)
