#!/usr/bin/env python3
"""
佳木斯市人民政府 - 公告公示 爬虫
CMS: 自定义政府CMS (新闻云平台)
API: /common/search/{channelId}?_isAgg=false&_isJson=true&_pageSize=15&_template=index&page={n}
数据: API直接返回完整JSON含标题/日期/正文HTML
页面: 259页×15条=3885条
"""

import sys, os, re, time, sqlite3, subprocess, hashlib, argparse
from datetime import datetime
from urllib.parse import urljoin
import requests

API_URL = "https://www.jms.gov.cn/common/search/cf5f6c0f46a24a73a31f5a8943cdddbe"
BASE_URL = "https://www.jms.gov.cn"
SITE_NAME = "佳木斯市_公告公示"
GROUP = "黑龙江"
DB_PATH = "/mnt/data/search.db"
PAGE_SIZE = 15
TOTAL = 3885
MAX_PAGES = 259  # 3885/15

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 "
                  "(KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def clean_content_html(html):
    """清理contentHtml: 去style/span/strong等无用标签，保留段落和链接"""
    if not html:
        return ""
    # 去掉style属性
    cleaned = re.sub(r'\sstyle="[^"]*"', '', html)
    # 扁平化span
    cleaned = re.sub(r'<span[^>]*>|</span>', '', cleaned)
    # 扁平化strong/b
    cleaned = re.sub(r'<strong[^>]*>|</strong>', '', cleaned)
    cleaned = re.sub(r'<b[^>]*>|</b>', '', cleaned)
    # 去掉font标签
    cleaned = re.sub(r'<font[^>]*>|</font>', '', cleaned)
    # 处理o:p标签
    cleaned = re.sub(r'<o:p[^>]*>|</o:p>', '', cleaned)
    # 保留p/table/a/img/br
    # 移除多余的空白行
    cleaned = re.sub(r'\n{3,}', '\n\n', cleaned)
    return cleaned.strip()


def crawl(max_pages=MAX_PAGES):
    """主爬取逻辑"""
    s = requests.Session()
    s.headers.update(HEADERS)

    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    # 检查已有page_url
    existing_cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    existing = set(row[0] for row in existing_cur.fetchall())

    total_new = total_skip = total_error = 0
    pages_to_crawl = min(max_pages, MAX_PAGES)
    print(f"[佳木斯·公告公示] 开始爬取, 已有 {len(existing)} 条, 目标 {pages_to_crawl} 页 (共{TOTAL}条)")

    for page in range(1, pages_to_crawl + 1):
        api_url = f"{API_URL}?_isAgg=false&_isJson=true&_pageSize={PAGE_SIZE}&_template=index&_rangeTimeGte=&_channelName=&page={page}"
        print(f"  第 {page}/{pages_to_crawl} 页", flush=True)

        try:
            resp = s.get(api_url, timeout=30)
            if resp.status_code != 200:
                print(f"    HTTP {resp.status_code}, 跳过")
                total_error += 1
                continue
            data = resp.json()
        except Exception as e:
            print(f"    API请求失败: {e}")
            total_error += 1
            time.sleep(3)
            continue

        results = data.get('data', {}).get('results', [])
        if not results:
            print("    无条目, 停止翻页")
            break

        for item in results:
            url = item.get('url', '')
            if not url:
                continue
            full_url = url if url.startswith('http') else urljoin(BASE_URL, url)
            if full_url in existing:
                total_skip += 1
                continue

            title = (item.get('title') or '').replace('&nbsp;', ' ').strip()
            if not title:
                title = (item.get('subTitle') or '').replace('&nbsp;', ' ').strip()

            # 日期
            publish_date = ""
            ds = item.get('publishedTimeStr', '')
            if ds and len(ds) >= 10:
                publish_date = ds[:10]

            # 正文
            content_html = item.get('contentHtml', '') or item.get('content', '')
            body = clean_content_html(content_html)

            # summary
            summary = re.sub(r'<[^>]+>', ' ', body)
            summary = re.sub(r'\s+', ' ', summary).strip()[:300]

            try:
                cur = db.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date,
                     summary, content, status, category, group_name, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                    SITE_NAME, full_url, full_url, title[:500], publish_date,
                    summary[:500], body, 'active', '', GROUP, ''
                ))
                if cur.rowcount > 0:
                    row = db.execute("SELECT id FROM gov_raw WHERE page_url=?", (full_url,)).fetchone()
                    if row:
                        rid = row[0]
                        esc_title = title.replace("'", "''")
                        esc_site = SITE_NAME.replace("'", "''")
                        esc_summary = summary.replace("'", "''")
                        sql = f"INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES({rid},'{esc_title}','{esc_site}','{esc_summary}')"
                        subprocess.run(["sqlite3", "-cmd", ".timeout 60000", DB_PATH, sql], capture_output=True, timeout=30)
                    existing.add(full_url)
                    db.commit()
                    total_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"    ✗ DB error: {e}")
                total_error += 1

        time.sleep(0.5)

    db.close()
    print(f"\n[佳木斯·公告公示] 完成！新增 {total_new}, 跳过 {total_skip}, 错误 {total_error}")
    return total_new


if __name__ == '__main__':
    parser = argparse.ArgumentParser(description='佳木斯市·公告公示爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES, help=f'最大页数 (默认 {MAX_PAGES})')
    args = parser.parse_args()
    crawl(max_pages=args.max_pages)
