#!/usr/bin/env python3
"""
crawl_xiuwen.py - 修文县人民政府-生态环境
https://www.xiuwen.gov.cn/zwgk_5667434/zdlygk_5874974/hjbh_5667505/
TRS WCM, div.mainbox正文, meta ArticleTitle/PubDate
"""
import os, re, sys, json, time, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE = "https://www.xiuwen.gov.cn"
LIST_BASE_URL = "/zwgk_5667434/zdlygk_5874974/hjbh_5667505"
LIST_URL = BASE + LIST_BASE_URL + "/index.html"
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
SITE_NAME = "修文县政府-生态环境"
CATEGORY = "环保公示"

HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}

print(f"[xiuwen] 3-year cutoff: {CUTOFF_DATE}")

def fetch(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"[xiuwen] fetch fail {url}: {e}")
        return ""

def parse_list(html):
    """Parse list items from TRS list page"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    # Find all <li> that contain <a> and <span>
    lis = soup.find_all('li')
    for li in lis:
        a = li.find('a')
        span = li.find('span')
        if a and span and a.get('href'):
            href = a['href']
            title = a.get_text(strip=True)
            date_str = span.get_text(strip=True)
            # Validate date format
            if re.match(r'\d{4}-\d{2}-\d{2}', date_str):
                items.append((title, href, date_str))
    return items

def get_detail(url):
    """Extract title, date, body from detail page"""
    html = fetch(url)
    if not html:
        return None, None, None
    
    soup = BeautifulSoup(html, 'html.parser')
    
    title = ''
    mt = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if mt and mt.get('content'):
        title = mt['content'].strip()
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)
    
    date_str = ''
    md = soup.find('meta', attrs={'name': 'PubDate'})
    if md and md.get('content'):
        m = re.search(r'(\d{4}-\d{2}-\d{2})', md['content'])
        if m:
            date_str = m.group(1)
    
    body = ''
    mainbox = soup.select_one('div.mainbox')
    if mainbox:
        body = str(mainbox)
    
    return title, date_str, body

# ── Get total pages from page 1 ──
html1 = fetch(LIST_URL)
items = parse_list(html1)
print(f"[xiuwen] Page 1: {len(items)} items")

# Find total pages from createPageHTML
total_pages = 1
m = re.search(r'createPageHTML\(parseInt\("(\d+)"\)', html1)
if m:
    total_pages = int(m.group(1))
print(f"[xiuwen] Total pages: {total_pages}")

# ── Collect all items from all pages ──
all_items = items
for page_idx in range(1, total_pages):
    page_url = BASE + LIST_BASE_URL + f"/index_{page_idx}.html"
    h = fetch(page_url)
    if h:
        page_items = parse_list(h)
        all_items.extend(page_items)
        print(f"[xiuwen] Page {page_idx+1}: {len(page_items)} items")
    time.sleep(0.5)

print(f"[xiuwen] Total items: {len(all_items)}")

# ── Filter by cutoff ──
active = [(t, u, d) for t, u, d in all_items if d >= CUTOFF_DATE]
skipped = len(all_items) - len(active)
print(f"[xiuwen] Within 3yr: {len(active)}, older: {skipped}")

# ── Fetch details ──
import sqlite3
conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()

new_count = dup_count = error_count = 0

for idx, (title, href, date_str) in enumerate(active, 1):
    # Build absolute URL
    if href.startswith("http"):
        detail_url = href
    elif href.startswith("./"):
        detail_url = BASE + LIST_BASE_URL + href[1:]  # ./202606/... -> /.../202606/...
    elif href.startswith("/"):
        detail_url = BASE + href
    elif href.startswith("../"):
        detail_url = BASE + "/" + href
    else:
        detail_url = BASE + LIST_BASE_URL + "/" + href
    
    try:
        cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
        if cur.fetchone():
            dup_count += 1
            continue
        
        d_title, d_date, body = get_detail(detail_url)
        if not d_title:
            d_title = title
        if not d_date:
            d_date = date_str
        
        date_rank = int(d_date.replace("-", "")) if d_date and "-" in d_date else 0
        summary = ""
        if body:
            text_soup = BeautifulSoup(body, 'html.parser')
            plain = text_soup.get_text(strip=True)
            summary = plain[:200]
        
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, d_title, detail_url, d_date, body, date_rank, CATEGORY, summary)
        )
        if cur.rowcount > 0:
            new_count += 1
    except Exception as e:
        error_count += 1
        print(f"  ERROR [{idx}] {title[:40]}: {e}")
    
    conn.commit()
    time.sleep(0.3)
    
    if idx % 20 == 0 or idx == len(active):
        print(f"  Progress {idx}/{len(active)}: +{new_count} new, {dup_count} dup, {error_count} err")

conn.close()
print(f"\n[xiuwen] Summary: +{new_count} new, {dup_count} dup, {error_count} err")
print(json.dumps({"site": SITE_NAME, "new": new_count, "dup": dup_count, "err": error_count, "total": len(active)}, ensure_ascii=False))
