#!/usr/bin/env python3
"""
九江经开区 - 生态环境 (jkq.jiujiang.gov.cn)
列表: /fdzdgk20251207/zdly/ggjg/sthj/
      index.html, index_1.html, index_2.html...
详情: /fdzdgk20251207/zdly/ggjg/sthj/202604/t20260429_7230813.html
正文: div.Zoom 或 div#Zoom
"""
import sys, os, re, time
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb, clean_html
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "九江经开区-生态环境"
BASE_URL = "https://jkq.jiujiang.gov.cn"
LIST_BASE = "https://jkq.jiujiang.gov.cn/fdzdgk20251207/zdly/ggjg/sthj/"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  fetch error: {e}")
        return None


def parse_list(html):
    """从列表页 HTML 提取所有条目 (title, url)"""
    items = []
    pattern = r'<a[^>]*href="([^"]*?\.html)"[^>]*>(.*?)</a>'
    for m in re.finditer(pattern, html, re.DOTALL):
        href = m.group(1).strip()
        title = re.sub(r"<[^>]+>", "", m.group(2)).strip()
        if not title or len(title) < 4:
            continue
        if any(kw in title for kw in ["首页", "末页", "下一页", "上一页", "尾页", "..."]):
            continue
        if href.startswith("/"):
            full_url = BASE_URL + href
        elif href.startswith("http"):
            full_url = href
        else:
            full_url = LIST_BASE.rstrip("/") + "/" + href.lstrip("./")
        items.append((title, full_url))
    return items


def get_page_url(page):
    """分页URL: index.html, index_1.html, index_2.html..."""
    if page == 1:
        return LIST_BASE + "index.html"
    return f"{LIST_BASE}index_{page - 1}.html"


def fetch_detail(url):
    """获取详情页: 标题、正文、发布日期"""
    html = fetch(url)
    if not html:
        return None, None, None

    result = {}

    # 标题: meta ArticleTitle 或 h1/h2
    m = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]*)"', html)
    if m:
        result["title"] = m.group(1).strip()
    if not result.get("title"):
        m = re.search(r"<h1[^>]*>(.*?)</h1>", html, re.DOTALL)
        if m:
            result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    if not result.get("title"):
        m = re.search(r"<h2[^>]*>(.*?)</h2>", html, re.DOTALL)
        if m:
            result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()

    # 日期: meta PubDate 或 "发布时间"
    m = re.search(r'<meta[^>]*name="PubDate"[^>]*content="(\d{4}-\d{1,2}-\d{1,2})', html)
    if m:
        result["publish_date"] = m.group(1)
    if not result.get("publish_date"):
        m = re.search(r"发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})", html)
        if m:
            result["publish_date"] = m.group(1)

    # 正文: div#Zoom 或 div.Zoom
    content = None
    m = re.search(r'<div[^>]*id="Zoom"[^>]*>(.*?)</div>\s*(?:<div|<!\-\-|$)', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    if not content or len(content) < 100:
        m = re.search(r'<div[^>]*class="Zoom"[^>]*>(.*?)</div>\s*(?:<div|<!\-\-|$)', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
    if not content or len(content) < 100:
        m = re.search(r'<div[^>]*class="[^"]*Zoom[^"]*"[^>]*>(.*?)</div>\s*(?:<div|<!\-\-|$)', html, re.DOTALL)
        if m:
            content = m.group(1).strip()

    if content:
        content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL | re.I)
        content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL | re.I)
        content = re.sub(r"<!--.*?-->", "", content, flags=re.DOTALL)
        if len(content) > 100:
            result["content"] = content

    # 兜底: 常见正文容器
    if not result.get("content"):
        for cls in ["TRS_Editor", "article-content", "content", "article", "text", "main"]:
            m = re.search(
                r'<(div|section)[^>]*class="[^"]*' + cls + r'[^"]*"[^>]*>(.*?)</\1>',
                html, re.DOTALL,
            )
            if m and len(m.group(2)) > 100:
                result["content"] = m.group(2).strip()
                break

    return result.get("title"), result.get("content"), result.get("publish_date")


def run(args=None):
    """主函数，支持 --incremental (args='1' 只跑第1页)"""
    incremental = bool(args)

    print(f"\n{'=' * 50}\n  {SITE_NAME}\n{'=' * 50}")

    all_list = []
    pages = 1 if incremental else MAX_PAGES

    for page in range(1, pages + 1):
        url = get_page_url(page)
        print(f"\n  第 {page} 页 [{url}]...", end=" ", flush=True)
        html = fetch(url)
        if not html:
            print("FAIL")
            if page == 1:
                alt_url = LIST_BASE
                print(f"  尝试备选 [{alt_url}]...", end=" ", flush=True)
                html = fetch(alt_url)
                if not html:
                    print("FAIL")
                    break
                print("OK")
            else:
                break
        items = parse_list(html)
        if not items:
            print("0 items")
            if page == 1:
                snippet = html[:2000].replace("\n", " ")
                print(f"  HTML预览: {snippet[:400]}...")
            break
        print(f"OK ({len(items)} items)")
        all_list.extend(items)

    print(f"\n  列表总计: {len(all_list)} 条（去重前）")

    # 去重 + 过滤
    seen = set()
    filtered = []
    skipped = 0
    for title, url in all_list:
        if url in seen:
            continue
        seen.add(url)
        if "项目" not in title:
            skipped += 1
            continue
        filtered.append((title, url))

    print(f"  过滤掉 {skipped} 条不含'项目'，剩余 {len(filtered)} 条")

    results = []
    for i, (title, url) in enumerate(filtered):
        print(f"  [{i + 1}/{len(filtered)}] {title[:50]}...", end=" ", flush=True)
        detail_title, content, pub_date = fetch_detail(url)
        if detail_title:
            title = detail_title
        if pub_date and pub_date < THREE_YEARS_AGO:
            print("SKIP (>3yr)")
            continue
        summary = re.sub(r"<[^>]+>", " ", content or "").strip()[:300]
        summary = re.sub(r"\s+", " ", summary)
        results.append({
            "site_name": SITE_NAME,
            "title": title,
            "url": url,
            "content": content or "",
            "pub_date": pub_date or "",
            "summary": summary,
            "tags": SITE_NAME,
        })
        print("OK")
        time.sleep(0.3)

    if results:
        push_to_searchdb(results, "jiujiang_jkq")
    else:
        print("  NO valid data for searchdb")

    print(f"\n  DONE: {len(results)} items stored")
    return results


if __name__ == "__main__":
    import argparse
    parser = argparse.ArgumentParser(description=f"Crawler: {SITE_NAME}")
    parser.add_argument("--incremental", action="store_true", help="just first page")
    args = parser.parse_args()
    t0 = time.time()
    run(args="1" if args.incremental else None)
    print(f"  Elapsed: {time.time() - t0:.1f}s")
