#!/usr/bin/env python3
"""
绵阳经济技术开发区管理委员会 - 公示公告
https://jkq.my.gov.cn/jkq/c102378/list.shtml
龙讯科技(Lonsun) CMS，JSON API + 详情页HTML
"""
import requests
import re
import sys
import os
import json
import time
from datetime import datetime, timedelta
from urllib.parse import urljoin

BASE_URL = "https://jkq.my.gov.cn"
API_URL = (BASE_URL + "/common/search/aeacd53099f94df0892fc7f44299a077"
           "?_isAgg=false&_isJson=true&_pageSize=20&_template=index&page=")
CHANNEL_ID = "aeacd53099f94df0892fc7f44299a077"
SITE_NAME = "jkq_my_gongshi"

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
})

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")


def fetch_list_page(page):
    """获取列表页JSON数据"""
    url = API_URL + str(page)
    try:
        r = session.get(url, timeout=30)
        r.encoding = "utf-8"
        data = r.json()
        results = data.get("data", {}).get("results", [])
        total = data.get("data", {}).get("total", 0)
        return results, total
    except Exception as e:
        print(f"  API page {page} fail: {e}")
        return [], 0


def parse_detail(html, url):
    """解析详情页提取正文内容"""
    result = {"content": "", "title": ""}

    # UCAPCONTENT标签内的内容
    m = re.search(r'<UCAPCONTENT[^>]*>(.*?)</UCAPCONTENT>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
        # 解析图片相对路径
        content = re.sub(
            r'src="(?!https?://)([^"]+)"',
            lambda m: 'src="' + urljoin(url, m.group(1)) + '"',
            content
        )
        # 解析附件相对路径
        content = re.sub(
            r'href="(?!https?://)([^"]+)"',
            lambda m: 'href="' + urljoin(url, m.group(1)) + '"',
            content
        )
        result["content"] = content

    # UCAPTITLE标签内的标题
    m = re.search(r'<UCAPTITLE[^>]*>(.*?)</UCAPTITLE>', html, re.DOTALL)
    if m:
        result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()

    return result


def insert_item(item, conn):
    """插入到数据库"""
    c = conn.cursor()

    title = item["title"]
    source_url = item["url"]
    content = item.get("content", "")
    pub_date = item.get("date", "")

    text_content = re.sub(r"<[^>]+>", "", content).strip() if content else ""

    if not content or len(text_content) < 50:
        print(f"  Short: {title[:30]}... ({len(text_content)} chars)")

    try:
        c.execute("""
            INSERT OR IGNORE INTO gov_raw
            (title, page_url, source_url, content, publish_date, site_name)
            VALUES (?, ?, ?, ?, ?, ?)
        """, (title, source_url, source_url, content, pub_date, SITE_NAME))
        affected = c.rowcount
        conn.commit()
        if affected > 0:
            print(f"  OK {title[:30]}...")
        return affected
    except Exception as e:
        print(f"  Insert fail: {e}")
        return 0


def crawl(days_back=365):
    import sqlite3
    now = datetime.now()
    cutoff = now - timedelta(days=days_back)

    conn = sqlite3.connect(SEARCH_DB, timeout=60)

    # 获取第1页，确定总页数
    print("Fetching page 1...")
    results, total = fetch_list_page(1)
    if not results:
        print("No data!")
        return

    page_size = 20
    total_pages = (total + page_size - 1) // page_size
    print(f"Total: {total} records, {total_pages} pages")

    all_items = []
    # 第1页数据
    for r in results:
        ts = r.get("publishedTime", 0) / 1000
        dt = datetime.fromtimestamp(ts) if ts else now
        if dt < cutoff and days_back < 3650:
            continue
        item = {
            "title": r.get("title", ""),
            "date": r.get("publishedTimeStr", "")[:10],
            "url": BASE_URL + r.get("url", ""),
            "manuscriptId": r.get("manuscriptId", ""),
            "content": r.get("content", "")
        }
        all_items.append(item)

    # 剩余页
    for page in range(2, total_pages + 1):
        print(f"\nFetching page {page}/{total_pages}...")
        results, _ = fetch_list_page(page)
        for r in results:
            ts = r.get("publishedTime", 0) / 1000
            dt = datetime.fromtimestamp(ts) if ts else now
            if dt < cutoff and days_back < 3650:
                continue
            item = {
                "title": r.get("title", ""),
                "date": r.get("publishedTimeStr", "")[:10],
                "url": BASE_URL + r.get("url", ""),
                "manuscriptId": r.get("manuscriptId", ""),
                "content": r.get("content", "")
            }
            all_items.append(item)
        time.sleep(0.5)

    print(f"\nTotal items to process: {len(all_items)}")
    total_ok = 0

    for i, item in enumerate(all_items):
        print(f"\n[{i+1}/{len(all_items)}] {item['title'][:30]}...")

        # Fetch detail page for full HTML content
        try:
            r = session.get(item["url"], timeout=30)
            r.encoding = "utf-8"
            detail = parse_detail(r.text, item["url"])
            if detail["content"]:
                item["content"] = detail["content"]
            if detail["title"]:
                item["title"] = detail["title"]
        except Exception as e:
            print(f"  Detail fail: {e}")

        affected = insert_item(item, conn)
        if affected > 0:
            total_ok += 1

        time.sleep(0.3)

    conn.close()
    print(f"\nDone: {total_ok} new, {len(all_items) - total_ok} skipped/dup")
    return total_ok


if __name__ == "__main__":
    days = int(sys.argv[1]) if len(sys.argv) > 1 else 365
    crawl(days_back=days)
