#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
沁源县人民政府 - 公示公告 爬虫
https://www.qinyuan.gov.cn/qyxxgk/zfxxgk/zfxxgkml/gsgg/
静态页面 index.html + index_N.html 分页, TRS_Editor 正文
"""

import json, re, sys, urllib.request, urllib.parse, ssl
from html import unescape

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

SITE_NAME = "沁源县公示公告"
BASE_URL = "https://www.qinyuan.gov.cn"
LIST_DIR = "/qyxxgk/zfxxgk/zfxxgkml/gsgg"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

MAX_PAGES = 5

def fetch_url(url, timeout=15):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=timeout, context=ssl_ctx)
    return resp.read().decode("utf-8", errors="replace")

def fetch_list_page(page_no):
    if page_no == 1:
        url = BASE_URL + LIST_DIR + "/index.html"
    else:
        url = BASE_URL + LIST_DIR + "/index_{}.html".format(page_no - 1)
    try:
        html = fetch_url(url)
    except Exception as e:
        print("  [ERROR] 第{}页请求失败: {}".format(page_no, e), file=sys.stderr)
        return []
    items = []
    for m in re.finditer(
        r'<li><a href="\./([^"]+)"[^>]*>([\s\S]*?)</a><span[^>]*>([^<]+)',
        html, re.DOTALL
    ):
        href = m.group(1).strip()
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        title = unescape(title)
        title = re.sub(r'\s+', ' ', title).strip()
        date = m.group(3).strip()

        url = BASE_URL + LIST_DIR + "/" + href
        items.append({"title": title, "url": url, "date": date})
    return items

def fetch_detail(url):
    try:
        html = fetch_url(url)
    except Exception as e:
        print("  [ERROR] 详情页失败: {} - {}".format(url, e), file=sys.stderr)
        return {"content": "", "attachments": []}

    # Content from TRS_Editor
    content = ""
    idx = html.find('<div class="article-con TRS_Editor">')
    if idx >= 0:
        idx_start = html.find(">", idx) + 1
        # Find the closing </div> - need to find the matching one
        depth = 1
        i = idx_start
        while i < len(html) and depth > 0:
            if html[i:i+6] == "</div>":
                depth -= 1
                if depth == 0:
                    break
            elif html[i:i+4] == "<div" and html[i+4] not in ">/":
                # might be <div...> opening
                m = re.match(r'<div[^>]*>', html[i:])
                if m and m.group(0) != html[i:i+4]:
                    depth += 1
                elif not m:
                    pass # skip non-matching
                elif m.group(0)[-2] != '/':
                    depth += 1
            elif html[i:i+3] == "<p " or html[i:i+2] == "<p>":
                pass
            i += 1
        if depth == 0:
            zoom_html = html[idx_start:i]
            content = extract_content(zoom_html)

        # Fallback: try trs_editor_view
        if not content:
            m = re.search(r'class="trs_editor_view[^"]*"[^>]*>([\s\S]*?)</div>\s*</div>', html, re.DOTALL)
            if m:
                content = extract_content(m.group(1))

    # Fallback: trs_editor_view directly
    if not content:
        m = re.search(r'<div[^>]*class="[^"]*TRS_Editor[^"]*"[^>]*>([\s\S]*?)</div>\s*</div>', html, re.DOTALL)
        if m:
            content = extract_content(m.group(1))

    # Attachments
    attachments = []
    for m in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', html, re.DOTALL | re.I):
        href = m.group(1)
        atext = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I) or \
           re.search(r'(/download|attached|upload)', href, re.I):
            if href.startswith("/"):
                href = BASE_URL + href
            elif not href.startswith("http"):
                href = url.rsplit("/", 1)[0] + "/" + href
            attachments.append({"name": atext, "url": href})

    return {"content": content, "attachments": attachments}

def embed_attachments_in_content(content, attachments):
    if not attachments:
        return content
    att_lines = ["[{}]({})".format(a["name"], a["url"]) for a in attachments]
    if content.strip():
        lines = [l.strip() for l in content.strip().split("\n")]
        all_filenames = all(re.search(r'\.\w{3,4}$', l) for l in lines if l)
        if all_filenames:
            return "\n".join(att_lines)
        else:
            return content + "\n\n" + "\n".join(att_lines)
    else:
        return "\n".join(att_lines)

def extract_content(html):
    parts = []
    for m in re.finditer(r'<p[^>]*>(.*?)</p\s*>|<table[^>]*>(.*?)</table\s*>|<img[^>]*src="([^"]+)"[^>]*>', html, re.DOTALL | re.I):
        tag = m.group(0)
        if tag.startswith("<p") or tag.startswith("<P"):
            p_text = re.sub(r'<[^>]+>', "", m.group(1) or "").strip()
            p_text = unescape(p_text).strip()
            if p_text and len(p_text) > 3 and not re.match(r'^[\s\xa0;]*$', p_text):
                parts.append(p_text)
        elif tag.startswith("<table") or tag.startswith("<TABLE"):
            table_md = html_table_to_markdown(m.group(0))
            if table_md:
                parts.append(table_md)
        elif tag.startswith("<img") or tag.startswith("<IMG"):
            src = m.group(3) or ""
            if src and not src.startswith("data:"):
                alt = re.search(r'alt="([^"]*)"', tag)
                alt_text = alt.group(1) if alt else ""
                if not src.startswith("http"):
                    src = BASE_URL + "/" + src.lstrip("/")
                parts.append("![{}]({})".format(alt_text, src))
    if not parts:
        text = re.sub(r'<[^>]+>', " ", html)
        text = unescape(text).strip()
        text = re.sub(r'\s+', " ", text)
        if text:
            parts.append(text)
    return "\n\n".join(parts)

def html_table_to_markdown(table_html):
    lines = []
    rows = re.findall(r'<tr[^>]*>(.*?)</tr\s*>', table_html, re.DOTALL | re.I)
    if not rows:
        return ""
    for ri, row in enumerate(rows):
        cells = re.findall(r'<t[dh][^>]*>(.*?)</t[dh]\s*>', row, re.DOTALL | re.I)
        cell_texts = []
        for c in cells:
            ct = re.sub(r'<[^>]+>', "", c).strip()
            ct = unescape(ct)
            ct = re.sub(r'\s+', " ", ct)
            cell_texts.append(ct)
        if ri == 0:
            lines.append("| " + " | ".join(cell_texts) + " |")
            lines.append("|---" * len(cell_texts) + "|")
        else:
            lines.append("| " + " | ".join(cell_texts) + " |")
    return "\n".join(lines)

def main():
    all_items = []
    for page_no in range(1, MAX_PAGES + 1):
        print("[INFO] 采集第{}页...".format(page_no), file=sys.stderr)
        items = fetch_list_page(page_no)
        if not items:
            print("[INFO] 第{}页无数据，结束".format(page_no), file=sys.stderr)
            break
        print("[INFO] 第{}页 {}条".format(page_no, len(items)), file=sys.stderr)
        for item in items:
            print("[INFO]  详情: {}...".format(item["title"][:30]), file=sys.stderr)
            detail = fetch_detail(item["url"])
            item["content"] = detail["content"]
            item["attachments"] = detail["attachments"]
            all_items.append(item)

    result = []
    for item in all_items:
        content = item.get("content", "")
        attachments = item.get("attachments", [])
        content = embed_attachments_in_content(content, attachments)

        att_str = ""
        if attachments:
            att_list = ["[{}]({})".format(a["name"], a["url"]) for a in attachments]
            att_str = "; ".join(att_list)
        result.append({
            "title": item["title"],
            "source_url": item["url"],
            "url": item["url"],
            "content": content,
            "site_name": SITE_NAME,
            "pub_date": item.get("date", ""),
            "summary": (item.get("content", "")[:200]),
            "attachments": att_str,
        })

    print(json.dumps(result, ensure_ascii=False, indent=2))

if __name__ == "__main__":
    main()
