#!/usr/bin/env python3
"""
砚山县人民政府 — 生态环境 (sthj)
================================
UCAP CMS, 数据嵌入在页面 JSON (listData.articleList)
列表: list.html → 提取 JSON 中的 articleList
详情: pc/content/content_{id}.html (div.zfxxgk_content)
"""
import sys, os, re, json, time, warnings
warnings.filterwarnings("ignore")
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
import requests
from bs4 import BeautifulSoup
import urllib.parse

SITE_NAME = "砚山县-生态环境"
BASE_URL = "https://www.yanshan.gov.cn/ysxrmzf/sthj/pc/list.html"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch(url, retries=3):
    for attempt in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, verify=False, timeout=30)
            r.encoding = "utf-8"
            return r.text
        except Exception as e:
            print("  [ERR] %s: %s" % (url[:80], e), file=sys.stderr)
            if attempt < retries - 1:
                time.sleep(2)
    return None

def parse_list_from_json(html):
    """从页面JSON提取文章列表（手动括号匹配）"""
    start = html.find("articleList: [")
    if start < 0:
        return []
    depth = 0
    i = start + len("articleList: [")
    arr_start = i - 1
    while i < len(html):
        ch = html[i]
        if ch == '[':
            depth += 1
        elif ch == ']':
            if depth == 0:
                arr_end = i + 1
                break
            depth -= 1
        i += 1
    else:
        return []
    raw = html[arr_start:arr_end]
    try:
        articles = json.loads(raw)
    except Exception:
        return []
    items = []
    for art in articles:
        title = art.get("showTitle") or art.get("title", "")
        title = re.sub(r"<[^>]+>", "", title).strip()
        pub_date = art.get("pubDate", "")[:10]
        urls = art.get("urls", {})
        if isinstance(urls, str):
            urls = json.loads(urls)
        detail_url = urls.get("pc", "")
        if detail_url and not detail_url.startswith("http"):
            detail_url = "https://www.yanshan.gov.cn" + detail_url
        if not detail_url:
            continue
        items.append((title, detail_url, pub_date))
    return items

def parse_detail(html):
    """解析详情页"""
    soup = BeautifulSoup(html, "html.parser")
    title_tag = soup.find("title")
    title = title_tag.get_text().strip() if title_tag else ""
    title = re.sub(r"\s*[-_].*$", "", title).strip()

    # Content container
    content_div = soup.select_one("div.text")
    if not content_div:
        content_div = soup.find("div", class_=lambda c: c and "content" in str(c) if c else False)
    if not content_div:
        return title, ""

    parts = []
    for tag in content_div.find_all(["p", "table", "img"], recursive=True):
        if tag.name == "p":
            text = tag.get_text(" ", strip=True)
            if text:
                parts.append(text)
        elif tag.name == "table":
            md = html_table_to_html(tag)
            if md:
                parts.append(md)
        elif tag.name == "img":
            src = tag.get("src", "")
            if src:
                if not src.startswith("http"):
                    src = "https://www.yanshan.gov.cn" + src
                parts.append("![%s](%s)" % (tag.get("alt", "图片"), src))
    return title, "\n\n".join(parts)

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=3)
    args = parser.parse_args()

    html = fetch(BASE_URL)
    if not html:
        print("Failed to fetch list page")
        return

    all_items = parse_list_from_json(html)
    print("Found %d items in JSON" % len(all_items))
    all_items = all_items[:args.pages * 20]  # 20 per page

    for idx, (title, url, date) in enumerate(all_items):
        print("[%d/%d] %s" % (idx+1, len(all_items), title[:50]))
        h = fetch(url)
        if not h:
            continue
        dt, content = parse_detail(h)
        if not content:
            print("  Empty, skip")
            continue
        rec = {
            "title": dt or title,
            "url": url,
            "source_url": url,
            "content": content,
            "pub_date": date,
            "site_name": SITE_NAME,
        }
        push_to_searchdb([rec])
        time.sleep(0.3)

    print("\nDone! %d items." % len(all_items))

if __name__ == "__main__":
    main()
