#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
爬虫：海盐县人民政府 - 通知公告
站点：https://www.haiyan.gov.cn/col/col1512827/
CMS：Hanweb JPAAS (桌面版)
API方式获取最新25条
"""

import requests, json, urllib.parse, re, sqlite3, os, time
from bs4 import BeautifulSoup

API_URL = "https://www.haiyan.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
API_PARAMS = {
    'parseType': 'bulidstatic', 'webId': '3018',
    'tplSetId': 'oiGRm428ZN4ykS3TEk2lW', 'pageType': 'column',
    'pageId': '1512827',
}
SITE_NAME = "海盐县-通知公告"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

def _get_db():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA synchronous=NORMAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn

def ensure_table(conn):
    conn.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT NOT NULL,
        content TEXT,
        source_url TEXT NOT NULL UNIQUE,
        publish_date TEXT,
        site_name TEXT,
        created_at TEXT DEFAULT (datetime('now','localtime'))
    )""")
    conn.execute("CREATE INDEX IF NOT EXISTS idx_gov_raw_site ON gov_raw(site_name)")
    conn.execute("CREATE INDEX IF NOT EXISTS idx_gov_raw_date ON gov_raw(publish_date)")
    conn.commit()

def row_exists(conn, source_url):
    cur = conn.execute("SELECT 1 FROM gov_raw WHERE source_url=?", (source_url,))
    return cur.fetchone() is not None

def insert_row(conn, title, content, source_url, publish_date):
    conn.execute("""INSERT OR IGNORE INTO gov_raw(title, content, source_url, publish_date, site_name)
        VALUES (?,?,?,?,?)""",
        (title, content, source_url, publish_date, SITE_NAME))
    if conn.total_changes > 0:
        return True
    return False

def extract_content_and_date(soup):
    """从详情页提取正文、标题和日期"""
    title = ""
    date = ""
    # 标题
    d = soup.select_one("div.main_lanmu")
    if d:
        title = d.get_text(strip=True)
    # 日期
    d = soup.select_one("div.bt_time")
    if d:
        txt = d.get_text(strip=True)
        m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
        if m:
            date = m.group(1)
    # Meta
    if not date:
        meta = soup.find('meta', attrs={'name': 'PubDate'})
        if meta and meta.get('content'):
            m = re.search(r'(\d{4}-\d{2}-\d{2})', meta['content'])
            if m:
                date = m.group(1)
    # 正文
    content_div = soup.select_one("div#zoom")
    if not content_div:
        content_div = soup.select_one("div.main_contert")
    if content_div:
        return str(content_div), title, date
    return "", title, date

def get_detail(url, session, retries=3):
    for attempt in range(retries):
        try:
            r = session.get(url, headers=HEADERS, timeout=60)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                soup = BeautifulSoup(r.text, 'html.parser')
                return extract_content_and_date(soup)
        except requests.RequestException as e:
            if attempt < retries - 1:
                time.sleep(2 ** attempt)
    return "", "", ""

def fetch_list():
    """通过JPAAS API获取最新文章列表"""
    qs = urllib.parse.urlencode(API_PARAMS)
    r = requests.get(API_URL + '?' + qs, headers=HEADERS, timeout=30)
    if r.status_code != 200:
        return []
    data = r.json()
    if not data.get('success'):
        return []
    html = data['data']['html']
    soup = BeautifulSoup(html, 'html.parser')
    results = []
    for a in soup.find_all('a', href=True):
        href = a['href']
        if '/col' in href and '/art/' in href:
            title = a.get_text(strip=True)
            if not title or len(title) < 5:
                continue
            url = href if href.startswith('http') else 'https://www.haiyan.gov.cn' + href
            # 找日期 - 在相邻的文本节点或span中
            date = ""
            parent = a.parent
            if parent:
                date_text = parent.get_text(strip=True)
                m = re.search(r'(\d{4}-\d{2}-\d{2})', date_text)
                if m:
                    date = m.group(1)
            results.append((title, url, date))
    return results

def crawl():
    conn = _get_db()
    ensure_table(conn)
    session = requests.Session()
    session.headers.update(HEADERS)

    articles = fetch_list()
    if not articles:
        print("未获取到文章列表")
        conn.close()
        return

    total_new = 0
    total_skip = 0
    for title, url, list_date in articles:
        if row_exists(conn, url):
            total_skip += 1
            continue
        content, detail_title, detail_date = get_detail(url, session)
        final_title = detail_title if detail_title else title
        final_date = detail_date if detail_date else list_date
        ct = BeautifulSoup(content or '', 'html.parser').get_text(strip=True)
        if len(ct) < 20:
            total_skip += 1
            continue
        if insert_row(conn, final_title, content, url, final_date):
            total_new += 1
            conn.commit()
            print("  ✓ [{}] {} ({})".format(total_new, final_title[:40], final_date))
        else:
            total_skip += 1

    conn.close()
    print("完成！新增{}条，跳过{}条".format(total_new, total_skip))

if __name__ == '__main__':
    crawl()
