#!/usr/bin/env python3
import os
"""
crawl_baokang.py - 保康县人民政府（多栏目）
原栏目: 生态环境 (http://www.baokang.gov.cn/xxgk/xxgkml/gysyjs/hjbh/)
新增: 基层政务-建设工程 (https://www.baokang.gov.cn/jcxxgk/bmxxgk/hbj/fdzdgk/qtzdgknr/jczwgk/)
TRS WCM, ul.info-list + createPageHTML pagination
正文: art_content + 附件区(shangnexe)
"""
import os, re, sys, time, json, urllib.request, urllib.error
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
HEADERS = {'User-Agent': 'Mozilla/5.0'}
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
print(f"[bk] 3-year cutoff: {CUTOFF}")

COLUMNS = [
    {
        "name": "保康县政府-生态环境",
        "base": "http://www.baokang.gov.cn/xxgk/xxgkml/gysyjs/hjbh",
        "list_url": "http://www.baokang.gov.cn/xxgk/xxgkml/gysyjs/hjbh/index.shtml",
        "category": "环保公示",
    },
    {
        "name": "保康县政府-基层政务公开",
        "base": "https://www.baokang.gov.cn/jcxxgk/bmxxgk/hbj/fdzdgk/qtzdgknr/jczwgk",
        "list_url": "https://www.baokang.gov.cn/jcxxgk/bmxxgk/hbj/fdzdgk/qtzdgknr/jczwgk/index.shtml",
        "category": "企业环保",
    },
]

def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=20).read().decode('utf-8', errors='replace')
    except Exception as e:
        print(f"[bk] fetch fail {url}: {e}")
        return ""

def get_detail(url):
    """Extract title, date, body from detail page.
    Body = art_content div + attachment section (shangnexe)."""
    html = fetch(url)
    if not html:
        return None, None, None
    soup = BeautifulSoup(html, 'html.parser')
    
    # title from meta tag
    title = ''
    mt = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if mt and mt.get('content'):
        title = mt['content'].strip()
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)
    
    # date from meta tag
    date_str = ''
    md = soup.find('meta', attrs={'name': 'PubDate'})
    if md and md.get('content'):
        m = re.search(r'(\d{4}-\d{2}-\d{2})', md['content'])
        if m:
            date_str = m.group(1)
    
    # Body: art_content + attachments
    body_parts = []
    
    art = soup.find('div', class_='art_content')
    if art:
        inner = ''.join(str(c) for c in art.children).strip()
        if inner:
            body_parts.append(inner)
    
    # Attachment section: shangnexe div (after art_content, before fxdy)
    att = soup.find('div', class_='shangnexe')
    if att:
        att_html = str(att).strip()
        if att_html:
            body_parts.append(att_html)
    
    body = '\n'.join(body_parts) if body_parts else ''
    
    return title, date_str, body

import sqlite3

results = []

for col in COLUMNS:
    site_name = col["name"]
    category = col["category"]
    list_url = col["list_url"]
    base_url = col["base"]
    
    print(f"\n{'='*60}")
    print(f"[bk] Processing: {site_name}")
    
    conn = sqlite3.connect(DB_PATH, timeout=30)
    cur = conn.cursor()
    
    html = fetch(list_url)
    if not html:
        print(f"[bk] FAILED to fetch list: {list_url}")
        results.append({"site": site_name, "error": "fetch list failed"})
        conn.close()
        continue
    
    page_match = re.search(r'createPageHTML\((\d+)', html)
    if not page_match:
        print(f"[bk] Cannot find page count!")
        total_pages = 1
    else:
        total_pages = int(page_match.group(1))
    print(f"[bk] Total pages: {total_pages}")
    
    total_new = 0
    total_skip = 0
    
    for page_idx in range(total_pages):
        if page_idx == 0:
            page_url = list_url
        else:
            page_url = base_url + f"/index_{page_idx}.shtml"
        
        html = fetch(page_url)
        if not html:
            continue
        
        soup = BeautifulSoup(html, 'html.parser')
        ul = soup.find('ul', class_='info-list')
        if not ul:
            print(f"[bk] No ul.info-list found on page {page_idx+1}")
            continue
        
        items = ul.find_all('li')
        print(f"[bk] Page {page_idx+1}/{total_pages}: {len(items)} items", end='')
        
        for li in items:
            a = li.find('a')
            if not a:
                continue
            
            href = a.get('href', '')
            title = a.get('title', a.get_text(strip=True))
            
            span = li.find('span')
            date_str = span.get_text(strip=True) if span else ''
            
            if date_str and date_str < CUTOFF:
                continue
            
            if href.startswith('//'):
                detail_url = 'https:' + href
            elif href.startswith('/'):
                detail_url = 'http://www.baokang.gov.cn' + href
            else:
                detail_url = href
            
            d_title, d_date, body = get_detail(detail_url)
            if not d_title:
                d_title = title
            if not d_date and date_str:
                d_date = date_str
            
            cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, site_name))
            if cur.fetchone():
                total_skip += 1
                continue
            
            body_clean = body if body else ''
            
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (title, content, page_url, site_name, publish_date) VALUES (?,?,?,?,?)",
                (d_title, body_clean, detail_url, site_name, d_date)
            )
            if cur.rowcount > 0:
                total_new += 1
                if total_new % 10 == 0:
                    conn.commit()
        
        print(f" → new:{total_new} skip:{total_skip}")
        conn.commit()
        time.sleep(0.3)
    
    conn.commit()
    conn.close()
    print(f"[bk] {site_name}: {total_new} new, {total_skip} existing skipped")
    results.append({"site": site_name, "new": total_new, "skip": total_skip})

print(f"\n{'='*60}")
print("Summary:")
for r in results:
    if "error" in r:
        print(f"  {r['site']}: ERROR - {r['error']}")
    else:
        print(f"  {r['site']}: +{r['new']} new, {r['skip']} skipped")
print(json.dumps(results, ensure_ascii=False))
