#!/usr/bin/env python3
"""
津市市人民政府 - 公示公告 爬虫
CMS: PowerCMS (static HTML pagination)
URL: https://www.jinshishi.gov.cn/zwxx/gsgg
"""

import sys
import os
import re
import time
import json
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import requests
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"

def get_connection():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

SITE_NAME = "津市市-公示公告"
BASE_URL = "https://www.jinshishi.gov.cn/zwxx/gsgg"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/130.0.0.0 Safari/537.36",
}
THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")


def crawl_list_page(page_num):
    """Crawl a single list page and return items"""
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"{BASE_URL}_{page_num}"
    
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")
    except Exception as e:
        print(f"[List] Error page {page_num}: {e}")
        return []
    
    ul = soup.find("ul", class_="newsList")
    if not ul:
        return []
    
    items = []
    for li in ul.find_all("li"):
        a = li.find("a")
        if not a or not a.get("href"):
            continue
        
        href = a["href"].strip()
        full_url = urljoin(BASE_URL, href)
        txt = li.get_text(strip=True)
        
        # Date is first 10 chars "YYYY-MM-DD"
        date_str = txt[:10] if re.match(r'\d{4}-\d{2}-\d{2}', txt[:10]) else ""
        title = txt[10:].strip() if date_str else txt
        
        items.append({
            "title": title,
            "url": full_url,
            "date_str": date_str,
        })
    
    return items


def crawl_detail(item):
    """Fetch a detail page and extract content"""
    url = item["url"]
    
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")
    except Exception as e:
        print(f"[Detail] Error {url}: {e}")
        return None
    
    # Title from meta
    title = item["title"]
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()
    
    # Date from meta
    date_str = item["date_str"]
    pd = soup.find("meta", attrs={"name": "PubDate"})
    if pd and pd.get("content"):
        date_str = pd["content"].strip()[:10]
    
    # Content from article.articleCon or div.printArea
    body_html = ""
    for sel in ["article.articleCon", "div.printArea"]:
        div = soup.select_one(sel)
        if div:
            body_html = str(div)
            break
    
    if not body_html:
        # Fallback: #content
        content_div = soup.select_one("#content")
        if content_div:
            body_html = str(content_div)
    
    # Summary
    summary_text = ""
    if body_html:
        txt_soup = BeautifulSoup(body_html, "html.parser")
        summary_text = txt_soup.get_text(separator=" ", strip=True)[:300]
    
    return {
        "site_name": SITE_NAME,
        "title": title,
        "page_url": url,
        "publish_date": date_str,
        "source_url": url,
        "content": body_html,
        "summary": summary_text,
    }


def save_to_db(records):
    conn = get_connection()
    cursor = conn.cursor()
    inserted = 0
    for rec in records:
        try:
            cursor.execute("""
                INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (
                rec["site_name"], rec["title"], rec["page_url"],
                rec["publish_date"], rec["source_url"], rec["content"],
                rec.get("summary", "")
            ))
            if cursor.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"[DB] Error: {e}")
    conn.commit()
    conn.close()
    return inserted


def main():
    print(f"=== {SITE_NAME} ===")
    print(f"3年截止: {THREE_YEARS_AGO}")
    
    # Step 1: Crawl list pages 1-43
    print("\n--- Step 1: List pages ---")
    all_items = []
    for p in range(1, 44):  # Pages 1-43 are within 3 years
        items = crawl_list_page(p)
        if not items:
            print(f"[List] Page {p}: empty, stopping")
            break
        # Filter by 3 years
        items = [it for it in items if it["date_str"] >= THREE_YEARS_AGO]
        all_items.extend(items)
        print(f"[List] Page {p}: {len(items)} items (total: {len(all_items)})")
        if items:
            print(f"  Oldest: {items[-1]['date_str']}")
        time.sleep(0.3)
    
    print(f"\nTotal items within 3 years: {len(all_items)}")
    
    if not all_items:
        print("No items, exiting")
        return
    
    # Step 2: Detail pages (multi-threaded)
    print(f"\n--- Step 2: {len(all_items)} detail pages ---")
    records = []
    errors = 0
    
    with ThreadPoolExecutor(max_workers=3) as executor:
        fut_map = {executor.submit(crawl_detail, it): it for it in all_items}
        for i, fut in enumerate(as_completed(fut_map)):
            rec = fut.result()
            if rec:
                records.append(rec)
            else:
                errors += 1
            if (i + 1) % 20 == 0:
                print(f"[Detail] {i+1}/{len(all_items)} done")
    
    print(f"\nDetail: {len(records)} success, {errors} errors")
    
    # Step 3: Save
    print("\n--- Step 3: Save to DB ---")
    inserted = save_to_db(records)
    print(f"Inserted {inserted} new (total fetched: {len(records)})")
    print("\n=== Done ===")


if __name__ == "__main__":
    main()
