#!/usr/bin/env python3
import os
"""
crawl_hljlanxi.py - 兰西县人民政府-通知公告
https://www.hljlanxi.gov.cn/lx/tzgg/tzgg.shtml
自定义CMS, API: /common/search/{channelId}, 1129条, 20条/页
"""
import os, re, sys, time, json, urllib.request, urllib.error
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
API_BASE = "https://www.hljlanxi.gov.cn/common/search/2e626efa8bc74387bfb3ab6d1fcdf175"
HEADERS = {'User-Agent': 'Mozilla/5.0'}

CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
print(f"[hlj] 3-year cutoff: {CUTOFF}")

def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=20).read().decode('utf-8', errors='replace')
    except Exception as e:
        print(f"[hlj] fetch fail {url}: {e}")
        return ""

def fetch_json(url):
    resp = fetch(url)
    if resp:
        try: return json.loads(resp)
        except: return None
    return None

def get_detail_body(url):
    """extract body from <div class="detail">"""
    html = fetch(url)
    if not html:
        return ''
    soup = BeautifulSoup(html, 'html.parser')
    body_el = soup.find('div', class_='detail')
    if body_el:
        return ''.join(str(c) for c in body_el.children).strip()
    return ''

# --- get total pages ---
api_url = f"{API_BASE}?_isAgg=true&_isJson=true&_pageSize=20&_template=index&_rangeTimeGte=&_channelName=&page=1"
data = fetch_json(api_url)
if not data or 'data' not in data:
    print("[hlj] Cannot fetch API data!")
    sys.exit(1)

total = data['data']['total']
rows = data['data']['rows']
total_pages = (total + rows - 1) // rows
print(f"[hlj] Total: {total}, Page size: {rows}, Pages: {total_pages}")

# --- crawl ---
import sqlite3
conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()

total_new = 0
total_skip = 0
total_old = 0

for page in range(1, min(total_pages + 1, 3)):  # max 2pg
    api_url = f"{API_BASE}?_isAgg=true&_isJson=true&_pageSize={rows}&_template=index&_rangeTimeGte=&_channelName=&page={page}"
    data = fetch_json(api_url)
    if not data or 'data' not in data:
        print(f"[hlj] Page {page}: fetch failed, stopping")
        break
    
    results = data['data'].get('results', [])
    if not results:
        print(f"[hlj] Page {page}: no results, stopping")
        break
    
    print(f"[hlj] Page {page}/{total_pages}: {len(results)} items", end='')
    
    for item in results:
        title = item.get('title', '')
        date_str = item.get('publishedTimeStr', '')[:10]
        raw_url = item.get('url', '')
        
        # cleanup URL
        detail_url = raw_url.replace('//lx', '/lx')
        if not detail_url.startswith('http'):
            detail_url = 'https://www.hljlanxi.gov.cn' + detail_url
        
        # date filter
        if date_str and date_str < CUTOFF:
            total_old += 1
            continue
        
        site_name = "兰西县-通知公告"
        
        # check existing
        cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, site_name))
        if cur.fetchone():
            total_skip += 1
            continue
        
        # get body
        body = get_detail_body(detail_url)
        
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw (title, content, page_url, site_name, publish_date) VALUES (?,?,?,?,?)",
            (title, body, detail_url, site_name, date_str)
        )
        if cur.rowcount > 0:
            total_new += 1
            if total_new % 30 == 0:
                conn.commit()
    
    print(f" → new:{total_new} skip:{total_skip} old:{total_old}")
    conn.commit()
    time.sleep(0.3)

conn.commit()
conn.close()
print(f"[hlj] DONE: {total_new} new, {total_skip} existing, {total_old} date-filtered")
