#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
盐亭县人民政府 - 经开区建设 爬虫
URL: https://www.yanting.gov.cn/yanting/c118277/list.shtml
CMS: 自定义 AJAX API驱动 (同 crawl_yanting.py c118536)
API: /common/search/{channelId}?page=N&_pageSize=20&_isAgg=true&_isJson=true&_template=index
列表JSON: {data: {total, results: [{title, url, publishedTimeStr}]}}
详情: <h1>标题, <div class="article-content">正文

用法:
  python3 crawl_yanting_jkq.py --pages=1     # 增量: 前1页
  python3 crawl_yanting_jkq.py --pages=5     # 首批前5页
"""
import sys, os, re, time, json, logging, sqlite3
from datetime import datetime, timedelta
from urllib.parse import urljoin

import requests
from bs4 import BeautifulSoup

def parse_pages(argv):
    if '--full' in argv:
        return 9999
    for i, a in enumerate(argv):
        if a == '--pages' and i + 1 < len(argv) and argv[i+1].isdigit():
            return int(argv[i+1])
        if a.startswith('--pages='):
            return int(a.split('=')[1])
        if a.isdigit():
            return int(a)
    return 1

MAX_PAGES = parse_pages(sys.argv[1:])
print(f"[yanting-jkq] max_pages={MAX_PAGES}")

DB_PATH = os.getenv("SEARCH_DB", "/mnt/data/search.db")
BASE_URL = "https://www.yanting.gov.cn"
SITE_NAME = "盐亭县-经开区建设"
CATEGORY = "经开区建设"
GROUP_NAME = "四川"
API_URL = f"{BASE_URL}/common/search/2769e005b69c4a48a96321df48ac10f6"
PAGE_SIZE = 20
DATE_CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
REQUEST_DELAY = 0.3

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0 Safari/537.36',
    'Accept': 'application/json',
}

logging.basicConfig(level=logging.INFO, format='[%(asctime)s] %(levelname)s %(message)s', datefmt='%H:%M:%S')
log = logging.getLogger(__name__)

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    conn.row_factory = sqlite3.Row
    return conn

def clean_title(t):
    if not t:
        return t
    t = t.replace("&middot;", "").replace("&nbsp;", "").replace("&ensp;", "").replace("&emsp;", "")
    t = re.sub(r"[\u200b\u200c\u200d\ufeff]", "", t)
    return t.strip()

def is_dup(conn, page_url):
    return conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone() is not None

def insert_item(conn, title, page_url, publish_date, content):
    date_rank = int(publish_date.replace("-", "")) if publish_date and "-" in publish_date else 0
    plain = ""
    if content:
        plain = BeautifulSoup(content, "html.parser").get_text(strip=True)[:200]
    try:
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary, script_name, group_name, has_table) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, title, page_url, publish_date, content, date_rank, CATEGORY, plain,
             "crawl_yanting_jkq.py", GROUP_NAME, 1 if "<table" in (content or "") else 0),
        )
        if cur.rowcount > 0:
            try:
                # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                conn.commit()
                conn.execute(
                    "INSERT OR REPLACE INTO gov_search(title, site_name, summary) VALUES (?, ?, ?)",
                    (title, SITE_NAME, plain),
                )
            except Exception as e:
                print(f"    [FTS WARN] {e}")
            return True
        return False
    except Exception as e:
        print(f"    [ERR] insert: {e}")
        return False

def fetch_json(url):
    try:
        r = requests.get(url, timeout=25, headers=HEADERS)
        r.encoding = 'utf-8'
        return r.json()
    except Exception as e:
        log.error(f"JSON请求失败 {url[:60]}: {e}")
        return None

def fetch(url):
    try:
        r = requests.get(url, timeout=25, headers={'User-Agent': HEADERS['User-Agent']})
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        log.error(f"请求失败 {url[:60]}: {e}")
        return None

def extract_content(node):
    """提取正文: 保留表格HTML、附件链接绝对化、段落分隔、去重复"""
    soup = BeautifulSoup(str(node), "html.parser")
    for tag in soup.find_all(["script", "style", "noscript"]):
        tag.decompose()
    for tag in soup.find_all(attrs={"style": re.compile(r"display\s*:\s*none", re.I)}):
        tag.decompose()
    for a in soup.find_all("a"):
        href = a.get("href")
        if href:
            a["href"] = urljoin(BASE_URL, href)
        img = a.find("img")
        if img:
            img.decompose()
    parts = []
    for el in soup.find_all(["p", "table", "div", "li", "h1", "h2", "h3", "h4"]):
        tag = el.name
        if tag == "tr":
            continue
        if tag == "p" and el.find_parent("table"):
            continue
        text = el.get_text(strip=True)
        if not text:
            continue
        if tag == "table":
            parts.append(str(el))
        elif tag in ("p", "div", "li", "h1", "h2", "h3", "h4"):
            if tag == "div" and el.find(["p", "table"]):
                continue
            parts.append(str(el))
    content = "\n\n".join(parts)
    segs = content.split("\n\n")
    deduped = []
    for s in segs:
        if not deduped or s != deduped[-1]:
            deduped.append(s)
    return "\n\n".join(deduped).strip()

def parse_detail(html, page_url):
    soup = BeautifulSoup(html, 'html.parser')
    h1 = soup.find('h1')
    title = clean_title(h1.get_text(strip=True)) if h1 else ''
    m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    pub_date = m.group(1) if m else ''
    content = ''
    content_div = soup.select_one('.article-content')
    if content_div:
        content = extract_content(content_div)
    return title, pub_date, content

def main():
    conn = get_conn()
    total_saved = total_skipped = total_err = 0

    data1 = fetch_json(f"{API_URL}?_isAgg=true&_isJson=true&_pageSize={PAGE_SIZE}&_template=index&page=1")
    if not data1 or not data1.get('data'):
        log.error("API请求失败")
        print("新增: 0")
        return

    total_records = data1['data'].get('total', 0)
    total_pages = (total_records + PAGE_SIZE - 1) // PAGE_SIZE
    pages = min(total_pages, MAX_PAGES)
    log.info(f"共 {total_records} 条, {total_pages} 页, 爬 {pages} 页")

    all_articles = []
    for item in data1['data'].get('results', []):
        all_articles.append({
            'title': item.get('title', ''),
            'page_url': urljoin(BASE_URL, item.get('url', '')),
            'publish_date': (item.get('publishedTimeStr') or '')[:10],
        })

    for page in range(2, pages + 1):
        time.sleep(REQUEST_DELAY)
        data = fetch_json(f"{API_URL}?_isAgg=true&_isJson=true&_pageSize={PAGE_SIZE}&_template=index&page={page}")
        if not data or not data.get('data'):
            break
        for item in data['data'].get('results', []):
            all_articles.append({
                'title': item.get('title', ''),
                'page_url': urljoin(BASE_URL, item.get('url', '')),
                'publish_date': (item.get('publishedTimeStr') or '')[:10],
            })

    log.info(f"共收集 {len(all_articles)} 条")

    for i, item in enumerate(all_articles):
        if item['publish_date'] and item['publish_date'] < DATE_CUTOFF:
            log.info(f"  跳过(日期过旧): {item['title'][:40]} ({item['publish_date']})")
            continue
        if is_dup(conn, item['page_url']):
            total_skipped += 1
            continue
        log.info(f"  详情 [{i+1}/{len(all_articles)}]: {item['title'][:50]}...")
        html = fetch(item['page_url'])
        if html:
            title, pub_date, content = parse_detail(html, item['page_url'])
            item['title'] = title or item['title']
            item['publish_date'] = pub_date or item['publish_date']
            if insert_item(conn, item['title'], item['page_url'], item['publish_date'], content):
                total_saved += 1
            else:
                total_skipped += 1
            conn.commit()
        time.sleep(REQUEST_DELAY)

    conn.close()
    log.info(f"完成: saved={total_saved}, skipped={total_skipped}, err={total_err}")
    print(f"新增: {total_saved}")

if __name__ == '__main__':
    main()
