#!/usr/bin/env python3
"""
crawl_longkou.py - 龙口市政府-通知公告
JCMS/JPAAS系统 + requests
"""
import os, re, sys, json, time, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import requests

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "龙口市政府-通知公告"
CATEGORY = "政府公告"
BASE = "https://www.longkou.gov.cn"
LIST_API = "/api-gateway/jpaas-publish-server/front/page/build/unit"
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
PAGE_SIZE = 50

BASE_PARAMS = {
    'parseType': 'bulidstatic',
    'webId': '153',
    'tplSetId': 'U1S5IC8QNYd71CkF9ENIL',
    'pageType': 'column',
    'tagId': '右侧列表',
    'editType': 'null',
    'pageId': '15019',
}

print("[longkou] 3-year cutoff: {}".format(CUTOFF_DATE))

# Get total count
resp = requests.get(BASE + LIST_API, params=BASE_PARAMS, headers=HEADERS, timeout=30)
data = resp.json()
html = data['data']['html']
count_match = re.search(r'count="(\d+)"', html)
total_count = int(count_match.group(1)) if count_match else 0
print("[longkou] Total items: {}".format(total_count))

total_pages = (total_count + PAGE_SIZE - 1) // PAGE_SIZE
print("[longkou] Total pages (size={}): {}".format(PAGE_SIZE, total_pages))

conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()
new_count = dup_count = error_count = 0
skipped_old = 0

for pg in range(1, total_pages + 1):
    params = dict(BASE_PARAMS)
    params['paramJson'] = json.dumps({"pageNo": pg, "pageSize": PAGE_SIZE})
    
    try:
        resp = requests.get(BASE + LIST_API, params=params, headers=HEADERS, timeout=30)
        data = resp.json()
        list_html = data['data']['html']
    except Exception as e:
        print("  [WARN] Page {} list error: {}".format(pg, e))
        error_count += PAGE_SIZE
        continue
    
    # Extract items: match ALL article links (both URL formats)
    soup = BeautifulSoup(list_html, 'html.parser')
    items = []
    for a in soup.find_all('a', href=True):
        href = a['href']
        # Match both formats: /col/col15019/art/... or /art/2025/.../art_15019_...
        if '15019' in href and 'art' in href:
            title = a.get('title', '') or a.get_text(strip=True)
            items.append((href, title))
    
    if not items:
        # fallback: look for any art_ links
        links = re.findall(r'<a[^>]+href="([^"]*art_[^"]+)"[^>]*title="([^"]*)"', list_html)
        for href, title in links:
            items.append((href, title))
    
    for href, title in sorted(set(items)):  # dedupe
        detail_url = href if href.startswith('http') else BASE + href
        
        # Year check from URL
        year_match = re.search(r'/art/(\d{4})', href)
        if year_match:
            year = int(year_match.group(1))
            if year < 2023:
                skipped_old += 1
                continue
            elif year == 2023:
                pass  # check exact date later
        
        # Check for existing
        cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
        if cur.fetchone():
            dup_count += 1
            continue
        
        # Fetch detail
        try:
            resp = requests.get(detail_url, headers=HEADERS, timeout=20)
            resp.encoding = 'utf-8'
            detail_html = resp.text
        except Exception as e:
            error_count += 1
            if error_count <= 3:
                print("  [WARN] Fetch error: {} -> {}".format(detail_url[:60], e))
            continue
        
        d_title = title
        d_date = ''
        body = ''
        
        mt = re.search(r'ArticleTitle"\s*content="([^"]+)"', detail_html)
        if mt and mt.group(1).strip():
            d_title = mt.group(1).strip()
        
        md = re.search(r'PubDate"\s*content="([^"]+)"', detail_html)
        if md:
            m = re.search(r'(\d{4}-\d{2}-\d{2})', md.group(1))
            if m:
                d_date = m.group(1)
        
        if d_date and d_date < CUTOFF_DATE:
            skipped_old += 1
            continue
        
        # Extract body
        zm = re.search(r'<div id="zoom"[^>]*>(.*?)</div>\s*<div class="other"', detail_html, re.DOTALL)
        if zm:
            body = zm.group(1).strip()
        else:
            zs = re.search(r'<div id="zoom"[^>]*>(.*?)</div>', detail_html, re.DOTALL)
            if zs:
                body = zs.group(1).strip()
        
        if not body:
            error_count += 1
            continue
        
        date_rank = int(d_date.replace("-", "")) if d_date else 0
        text_soup = BeautifulSoup(body, 'html.parser')
        summary = text_soup.get_text(strip=True)[:200]
        
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, d_title, detail_url, d_date, body, date_rank, CATEGORY, summary)
        )
        if cur.rowcount > 0:
            new_count += 1
    
    conn.commit()
    
    if pg % 10 == 0 or pg == total_pages:
        print("  Page {}/{}: +{} new, {} dup, {} err, {} old".format(pg, total_pages, new_count, dup_count, error_count, skipped_old))
    
    time.sleep(0.3)

conn.close()
print("\n[longkou] Summary: +{} new, {} dup, {} err, {} old".format(new_count, dup_count, error_count, skipped_old))
print(json.dumps({"site": SITE_NAME, "new": new_count, "dup": dup_count, "err": error_count}, ensure_ascii=False))
