#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
苍南县人民政府 - 环保审批公示 爬虫
https://www.cncn.gov.cn/col/col1229891308/index.html
JCMS系统，列表页通过API动态加载，详情页div#zoom正文
"""

import json
import re
import sys
import urllib.parse
import urllib.request
import ssl
from datetime import datetime
from html import unescape

# SSL context
ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

SITE_NAME = "苍南县环保审批公示"
BASE_URL = "https://www.cncn.gov.cn"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36"
}

# === Config ===
MAX_PAGES = 5  # 用户要求前5页

def fetch_url(url, timeout=15):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=timeout, context=ssl_ctx)
    return resp.read().decode("utf-8", errors="replace")

def fetch_list_page(page_no, page_size=10):
    """获取列表页API数据"""
    param_json = json.dumps({"pageNo": page_no, "pageSize": page_size}, ensure_ascii=False)
    params = {
        "parseType": "bulidstatic",
        "webId": "1831",
        "tplSetId": "Mm0JzuVkej5cUObgIOwPx",
        "pageType": "column",
        "tagId": "右侧信息列表",
        "editType": "null",
        "pageId": "1229891308",
        "paramJson": param_json,
    }
    qs = "&".join(k + "=" + urllib.parse.quote(v) for k, v in params.items())
    url = API_URL + "?" + qs
    try:
        html = fetch_url(url)
        data = json.loads(html)
        return data.get("data", {}).get("html", "")
    except Exception as e:
        print(f"  [ERROR] 列表页第{page_no}页请求失败: {e}", file=sys.stderr)
        return ""

def parse_list(html):
    """解析列表页提取文章信息"""
    items = []
    # 提取文章链接
    hrefs = re.findall(r'<a href="(/col/[^"/]+/art/\d+[^"]*\.html)" class="bt_link"', html)
    # 提取标题
    titles = re.findall(r'title="([^"]+)"', html)
    # 提取日期
    dates = re.findall(r'bt_time[^>]*>([^<]+)', html)

    n = min(len(hrefs), len(titles), len(dates))
    for i in range(n):
        href = hrefs[i]
        if href.startswith("/"):
            url = BASE_URL + href
        else:
            url = BASE_URL + "/" + href
        items.append({
            "title": titles[i],
            "url": url,
            "date": dates[i],
        })
    return items

def fetch_detail(url):
    """获取详情页内容"""
    try:
        html = fetch_url(url)
    except Exception as e:
        print(f"  [ERROR] 详情页请求失败: {url} - {e}", file=sys.stderr)
        return {"content": "", "attachments": []}

    # 标题
    title = ""
    m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m:
        title = re.sub(r'<[^>]+>', "", m.group(1)).strip()
        title = unescape(title)

    # 正文 - div#zoom
    content = ""
    idx = html.find('<div id="zoom">')
    if idx >= 0:
        end = html.find('</div>', idx)
        if end < 0:
            end = idx + 10000
        zoom_html = html[idx + len('<div id="zoom">'):end]
        content = extract_content(zoom_html)
    else:
        # fallback: try any div with zoom-related class
        for pat in [r'id="[^"]*zoom[^"]*"', r'class="[^"]*zoom[^"]*"']:
            m = re.search(pat, html)
            if m:
                idx = html.find(">", m.end())
                end = html.find("</div>", idx)
                if idx >= 0 and end > idx:
                    zoom_html = html[idx+1:end]
                    content = extract_content(zoom_html)
                    if content.strip():
                        break

    # 附件
    attachments = []
    for m in re.finditer(r'<a[^>]*href="([^"]*document/download[^"]*)"[^>]*>.*?</a>', html, re.DOTALL | re.I):
        href = m.group(1)
        if href.startswith("/"):
            href = BASE_URL + href
        atext = re.sub(r'<[^>]+>', "", m.group(0)).strip()
        attachments.append({
            "name": atext,
            "url": href
        })

    return {"content": content, "attachments": attachments}

def extract_content(html):
    """提取正文内容"""
    parts = []
    # 查找所有p, table, img
    for m in re.finditer(r'<p[^>]*>(.*?)</p\s*>|<table[^>]*>(.*?)</table\s*>|<img[^>]*src="([^"]+)"[^>]*>', html, re.DOTALL | re.I):
        tag = m.group(0)
        if tag.startswith("<p") or tag.startswith("<P"):
            p_text = re.sub(r'<[^>]+>', "", m.group(1) or m.group(0))
            p_text = unescape(p_text).strip()
            if p_text:
                # 跳过纯占位文本
                if not re.match(r'^[\s\xa0;]*$', p_text):
                    parts.append(p_text)
        elif tag.startswith("<table") or tag.startswith("<TABLE"):
            # 表格转Markdown
            table_md = html_table_to_markdown(m.group(0))
            if table_md:
                parts.append(table_md)
        elif tag.startswith("<img") or tag.startswith("<IMG"):
            src = m.group(3) or ""
            if src and not src.startswith("data:"):
                alt = re.search(r'alt="([^"]*)"', tag)
                alt_text = alt.group(1) if alt else ""
                if not src.startswith("http"):
                    src = BASE_URL + "/" + src.lstrip("/")
                parts.append(f"![{alt_text}]({src})")

    # 如果没有结构化的p/table/img，直接取文本
    if not parts:
        text = re.sub(r'<[^>]+>', " ", html)
        text = unescape(text).strip()
        text = re.sub(r'\s+', " ", text)
        if text:
            parts.append(text)

    return "\n\n".join(parts)

def html_table_to_markdown(table_html):
    """HTML表格转Markdown管道表"""
    lines = []
    rows = re.findall(r'<tr[^>]*>(.*?)</tr\s*>', table_html, re.DOTALL | re.I)
    if not rows:
        return ""

    for ri, row in enumerate(rows):
        cells = re.findall(r'<t[dh][^>]*>(.*?)</t[dh]\s*>', row, re.DOTALL | re.I)
        cell_texts = []
        colspan_found = False
        for c in cells:
            ct = re.sub(r'<[^>]+>', "", c).strip()
            ct = unescape(ct)
            ct = re.sub(r'\s+', " ", ct)
            # check for colspan
            if re.search(r'colspan\s*=\s*["\']?\d+', row, re.I):
                colspan_found = True
            cell_texts.append(ct)

        if colspan_found and ri == 0:
            lines.append("| " + " | ".join(cell_texts) + " |")
            lines.append("|---" * len(cell_texts) + "|")
        elif ri == 0:
            lines.append("| " + " | ".join(cell_texts) + " |")
            lines.append("|---" * len(cell_texts) + "|")
        else:
            lines.append("| " + " | ".join(cell_texts) + " |")

    return "\n".join(lines)

def main():
    all_items = []

    for page_no in range(1, MAX_PAGES + 1):
        print(f"[INFO] 正在采集第{page_no}页...", file=sys.stderr)
        html = fetch_list_page(page_no)
        if not html:
            break
        items = parse_list(html)
        if not items:
            print(f"[INFO] 第{page_no}页无数据，采集结束", file=sys.stderr)
            break
        print(f"[INFO] 第{page_no}页获取到{len(items)}条", file=sys.stderr)

        for item in items:
            print(f"[INFO]  详情: {item['title'][:30]}...", file=sys.stderr)
            detail = fetch_detail(item["url"])
            item["content"] = detail["content"]
            item["attachments"] = detail["attachments"]
            all_items.append(item)

    # 输出JSON供crawler_lib.push_to_searchdb处理
    result = []
    for item in all_items:
        attachments_str = ""
        if item.get("attachments"):
            att_list = []
            for a in item["attachments"]:
                att_list.append(f"[{a['name']}]({a['url']})")
            attachments_str = "; ".join(att_list)

        result.append({
            "title": item["title"],
            "source_url": item["url"],
            "url": item["url"],
            "content": item.get("content", ""),
            "site_name": SITE_NAME,
            "pub_date": item.get("date", ""),
            "summary": (item.get("content", "")[:200]),
            "attachments": attachments_str,
        })

    print(json.dumps(result, ensure_ascii=False, indent=2))

if __name__ == "__main__":
    main()
