#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
沁源县人民政府 - 重点领域信息公开 - 年度报告 爬虫
https://www.qinyuan.gov.cn/qyxxgk/zfxxgk/zfxxgkml/ggzy_235514/ndbg_235519/
"""

import json, re, sys, urllib.request, urllib.parse, ssl
from bs4 import BeautifulSoup
from html import unescape

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

SITE_NAME = "沁源县年度报告"
BASE_URL = "https://www.qinyuan.gov.cn"
LIST_DIR = "/qyxxgk/zfxxgk/zfxxgkml/ggzy_235514/ndbg_235519"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
MAX_PAGES = 5

def fetch_url(url, timeout=15):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=timeout, context=ssl_ctx)
    return resp.read().decode("utf-8", errors="replace")

def fetch_list_page(page_no):
    if page_no == 1:
        url = BASE_URL + LIST_DIR + "/index.html"
    else:
        url = BASE_URL + LIST_DIR + "/index_{}.html".format(page_no - 1)
    try:
        html = fetch_url(url)
    except Exception as e:
        print("  [ERROR] page {}: {}".format(page_no, e), file=sys.stderr)
        return []
    items = []
    for m in re.finditer(
        r'<a href="([^"]+)"[^>]*title="([^"]*)"[^>]*>([\s\S]*?)</a><span[^>]*>([^<]*)</span>',
        html, re.DOTALL
    ):
        href = m.group(1).strip()
        title = m.group(2).strip()
        date_raw = m.group(4).strip()
        date_raw = date_raw.replace("&nbsp;", "").strip()
        m2 = re.search(r"(\d{4}-\d{2}-\d{2})", date_raw)
        date = m2.group(1) if m2 else ""
        full_url = urllib.parse.urljoin(url, href)
        items.append({"title": title, "url": full_url, "date": date})
    return items

def fetch_detail(url):
    try:
        html = fetch_url(url)
    except Exception as e:
        print("  [ERROR] detail: {} - {}".format(url, e), file=sys.stderr)
        return {"content": "", "attachments": []}

    content = ""
    idx = html.find('<div class="article-con TRS_Editor">')
    if idx >= 0:
        idx_start = html.find(">", idx) + 1
        depth = 1
        i = idx_start
        while i < len(html) and depth > 0:
            if html[i:i+6] == "</div>":
                depth -= 1
            elif html[i:i+4] == "<div" and html[i+4] not in ">/" and html[i+5:i+6] not in ">/":
                open_m = re.match(r'<div[^>]*>', html[i:])
                if open_m and not open_m.group(0).endswith("/>"):
                    depth += 1
            i += 1
        if depth == 0:
            zoom_html = html[idx_start:i]
            content = extract_content(zoom_html)

    if not content:
        m = re.search(r'<div[^>]*class="[^"]*article-con[^"]*"[^>]*>([\s\S]*?)</div>', html, re.DOTALL)
        if m:
            content = extract_content(m.group(1))

    attachments = []
    for m in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', html, re.DOTALL | re.I):
        href = m.group(1)
        atext = re.sub(r'<[^>]+>', "", m.group(2)).strip()
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I) or \
           re.search(r'(/download|attached|upload)', href, re.I):
            href_full = urllib.parse.urljoin(url, href)
            attachments.append({"name": atext, "url": href_full})
    return {"content": content, "attachments": attachments}

def embed_attachments_in_content(content, attachments):
    """将附件链接嵌入正文"""
    if not attachments:
        return content
    att_lines = []
    for a in attachments:
        att_lines.append("[{}]({})".format(a["name"], a["url"]))
    # Check if content is basically just filenames
    if content.strip():
        # Check if content consists only of filenames
        lines = [l.strip() for l in content.strip().split("\n")]
        all_filenames = all(re.search(r'\.\w{3,4}$', l) for l in lines if l)
        if all_filenames:
            return "\n".join(att_lines)
        else:
            return content + "\n\n" + "\n".join(att_lines)
    else:
        return "\n".join(att_lines)

def extract_content(html):
    parts = []
    for m in re.finditer(r'<p[^>]*>(.*?)</p\s*>|<table[^>]*>(.*?)</table\s*>|<img[^>]*src="([^"]+)"[^>]*>', html, re.DOTALL | re.I):
        tag = m.group(0)
        if tag.startswith("<p") or tag.startswith("<P"):
            p_text = re.sub(r'<[^>]+>', "", m.group(1) or "").strip()
            p_text = unescape(p_text).strip()
            if p_text and len(p_text) > 3 and not re.match(r'^[\s\xa0;]*$', p_text):
                parts.append(p_text)
        elif tag.startswith("<table") or tag.startswith("<TABLE"):
            table_md = html_table_to_html(m.group(0))
            if table_md:
                parts.append(table_md)
        elif tag.startswith("<img") or tag.startswith("<IMG"):
            src = m.group(3) or ""
            if src and not src.startswith("data:"):
                alt = re.search(r'alt="([^"]*)"', tag)
                alt_text = alt.group(1) if alt else ""
                if not src.startswith("http"):
                    src = urllib.parse.urljoin(BASE_URL, src)
                parts.append("![{}]({})".format(alt_text, src))
    if not parts:
        text = re.sub(r'<[^>]+>', " ", html)
        text = unescape(text).strip()
        text = re.sub(r'\s+', " ", text)
        if text:
            parts.append(text)
    return "\n\n".join(parts)

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def main():
    all_items = []
    for page_no in range(1, MAX_PAGES + 1):
        print("[INFO] Page {}".format(page_no), file=sys.stderr)
        items = fetch_list_page(page_no)
        if not items:
            break
        print("[INFO] Page {}: {} items".format(page_no, len(items)), file=sys.stderr)
        for item in items:
            print("[INFO]  {}...".format(item["title"][:30]), file=sys.stderr)
            detail = fetch_detail(item["url"])
            item["content"] = detail["content"]
            item["attachments"] = detail["attachments"]
            all_items.append(item)

    result = []
    for item in all_items:
        content = item.get("content", "")
        attachments = item.get("attachments", [])
        content = embed_attachments_in_content(content, attachments)

        att_str = ""
        if attachments:
            att_list = ["[{}]({})".format(a["name"], a["url"]) for a in attachments]
            att_str = "; ".join(att_list)
        result.append({
            "title": item["title"],
            "source_url": item["url"],
            "url": item["url"],
            "content": content,
            "site_name": SITE_NAME,
            "pub_date": item.get("date", ""),
            "summary": (item.get("content", "")[:200]),
            "attachments": att_str,
        })
    print(json.dumps(result, ensure_ascii=False, indent=2))

if __name__ == "__main__":
    main()
