#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
海宁市人民政府信息公开平台 爬虫 (JCMS/JPAAS)
站点: www.haining.gov.cn
栏目: col1455897 - 重点领域信息公开 > 环境保护 (E1-2)
API返回全量列表HTML，逐条抓取详情
"""

import sys, os, re, json, sqlite3, warnings
warnings.filterwarnings('ignore')
from datetime import datetime

# ─── 将 /root/gov_crawler 加入 sys.path 以导入 crawler_lib ───
_HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, _HERE)
sys.path.insert(0, os.path.join(_HERE, '..', 'crawler'))

from crawler_lib import fetch_page

# ─── 常量 ───
SITE_NAME = "海宁市-重点领域-环境保护"
DOMAIN = "www.haining.gov.cn"
BASE_URL = "https://www.haining.gov.cn"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

API_URL = (
    "https://www.haining.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
    "?parseType=bulidstatic&webId=2780&tplSetId=oo54alAKysOEKkUzWqYCa"
    "&pageType=column&tagId=%E7%BB%84%E9%85%8D%E5%88%86%E7%B1%BBlist"
    "&editType=null&pageId=1455897"
    "&paramJson=%7B%22pageNo%22%3A1%2C%22pageSize%22%3A99999%2C%22search%22%3A%22%7B%5C%22xxgkId%5C%22%3A%5C%22E1-2%5C%22%2C%5C%22xxgkType%5C%22%3A%5C%22xxgk_combination%5C%22%2C%5C%22className%5C%22%3A%5C%22%5C%22%7D%22%7D"
)


# ─── DB 写入 ───
def insert_db(items):
    """直接连接 /root/search.db，写入 gov_raw + 同步 FTS5"""
    if not items:
        return
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA busy_timeout=8000")
    db.execute("PRAGMA synchronous=NORMAL")

    # 确保表存在
    db.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        site_name TEXT, source_url TEXT UNIQUE, page_url TEXT,
        title TEXT, publish_date TEXT, summary TEXT,
        content TEXT, status TEXT, category TEXT, tags TEXT,
        attachments TEXT
    )""")
    db.execute("""CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(
        title, site_name, summary, content='gov_raw', content_rowid='id'
    )""")
    db.commit()

    ok = 0
    skip = 0
    for item in items:
        try:
            title = (item.get("title") or "")[:500]
            summary = (item.get("summary") or title)[:500]
            content = item.get("content") or ""
            pub_date = (item.get("publish_date") or "")[:10]
            source_url = item.get("source_url") or ""
            page_url = item.get("url") or source_url
            attachments = item.get("attachments") or ""

            db.execute("""INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date,
                 summary, content, status, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                SITE_NAME, source_url, page_url, title, pub_date,
                summary, content, "active", attachments
            ))
            if db.total_changes > 0:
                ok += 1
            else:
                skip += 1
        except Exception:
            skip += 1

    db.commit()

    # 同步 FTS
    try:
        db.execute("""INSERT INTO gov_search(rowid, title, site_name, summary)
            SELECT r.id, r.title, r.site_name, r.summary
            FROM gov_raw r
            WHERE r.site_name=? AND r.id NOT IN (SELECT rowid FROM gov_search)""",
            (SITE_NAME,))
        db.commit()
    except Exception as e:
        print(f"  [FTS ERROR] {e}")

    db.close()
    print(f"  [DB] 新增: {ok}, 跳过: {skip}")
    return ok


# ─── 正文清洗 ───
def clean_content(html):
    """从 HTML 中提取纯文本，去掉 <style>/<script>，保留表格"""
    if not html:
        return ""
    # 去掉 <style>...</style>
    html = re.sub(r'<style[^>]*>.*?</style>', '', html, flags=re.DOTALL | re.IGNORECASE)
    # 去掉 <script>...</script>
    html = re.sub(r'<script[^>]*>.*?</script>', '', html, flags=re.DOTALL | re.IGNORECASE)
    # 去掉注释
    html = re.sub(r'<!--.*?-->', '', html, flags=re.DOTALL)
    return html.strip()


def extract_text_content(html):
    """将清洗后的 HTML 转为纯文本（但保留表格 HTML 标签占位）"""
    if not html:
        return ""
    # 保留表格，其它标签转为文本
    text = html
    # 将 <br>, </p>, </div>, </li>, </tr>, </th>, </td> 换行
    text = re.sub(r'<br\s*/?>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</p>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</div>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</li>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</tr>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</th>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</td>', '\n', text, flags=re.IGNORECASE)
    # 去掉其他 HTML 标签（保留表格标签）
    text = re.sub(r'<(?!/?table|/?tr|/?td|/?th|/?thead|/?tbody|/?tfoot|/?caption|/?colgroup|/?col)[^>]*>', '', text)
    # 清理空白
    text = re.sub(r'&nbsp;', ' ', text)
    text = re.sub(r'&lt;', '<', text)
    text = re.sub(r'&gt;', '>', text)
    text = re.sub(r'&amp;', '&', text)
    text = re.sub(r'\n{3,}', '\n\n', text)
    text = re.sub(r'[ \t]+', ' ', text)
    return text.strip()


# ─── 附件提取 ───
def extract_attachments(html, base_url=BASE_URL):
    """从 HTML 中提取附件链接，返回附件描述文本"""
    if not html:
        return ""
    ext_pattern = r'\.(doc|docx|pdf|xls|xlsx|rar|zip)(\?[^\s"\'<>]*)?'
    links = re.findall(
        r'<a[^>]*href=["\']([^"\']*?(?:' + ext_pattern + r'))["\'][^>]*>([^<]*)</a>',
        html, re.IGNORECASE
    )
    if not links:
        return ""
    parts = []
    for href, _, _, text in links:
        full_url = href if href.startswith('http') else (base_url.rstrip('/') + '/' + href.lstrip('/'))
        parts.append(f"[附件: {text.strip()}]({full_url})")
    return "\n".join(parts)


# ─── 解析 API 返回的列表 HTML ───
def parse_list_html(html):
    """从 API 返回的 json['data']['html'] 中提取文章列表"""
    items = []
    if not html:
        return items

    # 提取所有 <li class="cf"> 条目
    # 使用正则匹配
    pattern = re.compile(
        r'<li\s+class=["\']cf["\'][^>]*>'
        r'.*?<a\s+class=["\']fl["\'][^>]*href=["\']([^"\']+)["\'][^>]*title=["\']([^"\']*)["\'][^>]*>.*?</a>'
        r'.*?<span\s+class=["\']fr["\'][^>]*>([^<]*)</span>'
        r'.*?</li>',
        re.DOTALL | re.IGNORECASE
    )
    for m in pattern.finditer(html):
        href = m.group(1).strip()
        title = m.group(2).strip()
        date_str = m.group(3).strip()
        if not href or not title:
            continue
        page_url = BASE_URL.rstrip('/') + '/' + href.lstrip('/') if not href.startswith('http') else href
        source_url = page_url  # source_url 也使用完整URL
        items.append({
            "href": href,
            "page_url": page_url,
            "source_url": page_url,
            "title": title,
            "date": date_str,
        })
    return items


# ─── 获取详情 ───
def fetch_detail(item):
    """获取详情页 div.article 内容"""
    url = item["page_url"]
    html = fetch_page(url, timeout=30)
    if not html:
        print(f"  ⚠ 详情页获取失败: {url}")
        return None

    # ★ 从详情页提取完整标题（API 列表返回的标题被截断为 36字+...）
    detail_title = None
    # 优先从 <meta ArticleTitle> 取
    mt = re.search(r'<meta\s+name=["\']ArticleTitle["\'][^>]*content=["\']([^"\']+)["\']', html, re.IGNORECASE)
    if mt:
        detail_title = mt.group(1).strip()
    else:
        # 从 <div class="title"> 提取
        dt = re.search(r'<div[^>]*class=["\']title["\'][^>]*>\s*([^<]+)\s*</div>', html, re.DOTALL)
        if dt:
            detail_title = dt.group(1).strip()
        else:
            # 从 <title> 标签提取
            tt = re.search(r'<title>([^<]+)</title>', html, re.IGNORECASE)
            if tt:
                detail_title = tt.group(1).strip()

    # 提取 div.article 内容
    m = re.search(r'<div[^>]*class=["\']article["\'][^>]*>(.*?)</div>\s*(?:<|$)', html, re.DOTALL | re.IGNORECASE)
    if not m:
        # 尝试其他常见容器
        m = re.search(r'<div[^>]*class=["\'](?:content|main|text|TRS_Editor)["\'](?:[^>]*class=["\'][^>]*)?>(.*?)</div>\s*(?:<|$)', html, re.DOTALL | re.IGNORECASE)
    if not m:
        print(f"  ⚠ 未找到内容容器: {url}")
        return None

    raw_html = m.group(1)
    # 清洗
    cleaned = clean_content(raw_html)
    # 提取附件
    attachments = extract_attachments(raw_html)
    # 提取文本内容（保留表格）
    text_content = extract_text_content(cleaned)

    # 如果有附件，嵌入正文末尾
    if attachments:
        text_content = text_content.rstrip() + "\n\n--- 附件 ---\n" + attachments

    return {
        "title": detail_title,
        "content": text_content,
        "attachments": attachments,
    }


# ─── 标准化日期 ───
def normalize_date(date_str):
    """将各种格式的日期转为 YYYY-MM-DD"""
    if not date_str:
        return ""
    date_str = date_str.strip()
    # 2026-07-16
    m = re.match(r'(\d{4})-(\d{1,2})-(\d{1,2})', date_str)
    if m:
        y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
        if 1990 <= y <= 2099 and 1 <= mo <= 12 and 1 <= d <= 31:
            return f"{y:04d}-{mo:02d}-{d:02d}"
    # 2026/07/16
    m = re.match(r'(\d{4})/(\d{1,2})/(\d{1,2})', date_str)
    if m:
        y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
        if 1990 <= y <= 2099 and 1 <= mo <= 12 and 1 <= d <= 31:
            return f"{y:04d}-{mo:02d}-{d:02d}"
    # 2026年07月16日
    m = re.match(r'(\d{4})年(\d{1,2})月(\d{1,2})日', date_str)
    if m:
        y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
        if 1990 <= y <= 2099 and 1 <= mo <= 12 and 1 <= d <= 31:
            return f"{y:04d}-{mo:02d}-{d:02d}"
    # YYYYMMDD
    m = re.match(r'(\d{4})(\d{2})(\d{2})', date_str)
    if m:
        y, mo, d = int(m.group(1)), int(m.group(2)), int(m.group(3))
        if 1990 <= y <= 2099 and 1 <= mo <= 12 and 1 <= d <= 31:
            return f"{y:04d}-{mo:02d}-{d:02d}"
    return date_str[:10]


# ─── 主入口 ───
def run(max_items=None):
    print(f"\n{'='*60}")
    print(f"🚀 {SITE_NAME}")
    print(f"{'='*60}")

    # 1. 调用 API 获取全量列表
    print("\n📥 获取列表...")
    api_resp = fetch_page(API_URL, timeout=60)
    if not api_resp:
        print("❌ API 请求失败")
        return

    try:
        data = json.loads(api_resp)
    except json.JSONDecodeError as e:
        print(f"❌ JSON 解析失败: {e}")
        return

    list_html = data.get("data", {}).get("html")
    if not list_html:
        print("❌ 未找到 data.html")
        print(f"   API响应前200字: {api_resp[:200]}")
        return

    # 2. 解析列表
    all_items = parse_list_html(list_html)
    print(f"   API返回: {len(all_items)} 条")

    if not all_items:
        print("❌ 列表为空，检查解析逻辑")
        return

    # 3. 按 --pages 限制
    if max_items and max_items > 0 and max_items < len(all_items):
        all_items = all_items[:max_items]
        print(f"   --pages 限制: 取前 {max_items} 条")

    # 4. 逐条抓取详情
    success = 0
    fail = 0
    db_items = []

    for i, item in enumerate(all_items, 1):
        print(f"\n[{i}/{len(all_items)}] {item['title'][:50]}...")
        print(f"   日期: {item['date']}   URL: {item['page_url'][:60]}...")

        detail = fetch_detail(item)
        if detail is None:
            fail += 1
            continue

        pub_date = normalize_date(item["date"])
        # 使用详情页完整标题（API 列表标题被截断）
        db_title = detail.get("title") or item["title"]
        summary = db_title[:200]
        content_text = detail["content"]

        db_items.append({
            "title": db_title,
            "source_url": item["source_url"],
            "url": item["page_url"],
            "publish_date": pub_date,
            "summary": summary,
            "content": content_text,
            "attachments": detail["attachments"],
        })
        success += 1

        print(f"   ✅ 正文长度: {len(content_text)} 字")

    # 5. 写入 DB
    print(f"\n{'='*60}")
    print(f"📦 写入数据库... 成功: {success}, 失败: {fail}")
    if db_items:
        insert_db(db_items)
    print(f"{'='*60}")
    print(f"✅ 完成! 共处理 {len(all_items)} 条，成功 {success}，失败 {fail}")


if __name__ == "__main__":
    # 解析 --pages N 参数
    max_pages = None
    args = sys.argv[1:]
    for i, arg in enumerate(args):
        if arg == "--pages" and i + 1 < len(args):
            try:
                max_pages = int(args[i + 1])
            except ValueError:
                pass
    # 也支持直接传数字（兼容旧模式）
    if max_pages is None:
        for arg in args:
            if arg.lstrip('-').isdigit():
                max_pages = int(arg.lstrip('-'))
                break

    run(max_items=max_pages)
