#!/usr/bin/env python3
"""
爬虫: 上虞区人民政府 - 建设项目环境影响评价信息公示
站点: www.shangyu.gov.cn (JCMS)
"""

import requests, re, json, os, sys, time
from datetime import datetime

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "绍兴市上虞区人民政府"
LINK_FILE = "/tmp/shangyu_links.txt"

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
})

import sqlite3
conn = sqlite3.connect(DB_PATH)
c = conn.cursor()

def load_links():
    with open(LINK_FILE, "r") as f:
        return [line.strip() for line in f if line.strip()]

def fetch_detail(url):
    try:
        r = session.get(url, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        return {"title": "", "date": "", "body": "", "error": str(e)}
    
    title = ""
    m = re.search(r'<h2[^>]*>(.*?)</h2>', html, re.DOTALL)
    if m:
        title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    if not title:
        m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
        if m:
            title = m.group(1).strip()
    
    date = ""
    m = re.search(r'<meta[^>]*name=["\']PubDate["\'][^>]*content=["\']([^"\']+)["\']', html)
    if m:
        date = m.group(1).strip()
    if not date:
        m = re.search(r'(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}', html)
        if m:
            date = m.group(1)
    
    body = ""
    m = re.search(r'<div[^>]*id=["\']zoom["\'][^>]*>(.*?)</div>\s*<div[^>]*class=["\']bt_xx["\']', html, re.DOTALL)
    if m:
        body = m.group(1).strip()
    if not body:
        m = re.search(r'<div[^>]*id=["\']zoom["\'][^>]*>(.*?)</div>', html, re.DOTALL)
        if m:
            body = m.group(1).strip()
    
    return {"title": title, "date": date, "body": body, "error": None}

def crawl():
    print(f"[{datetime.now().strftime('%Y-%m-%d %H:%M:%S')}] Starting shangyu EIA crawler", flush=True)
    
    # Load links
    all_links = load_links()
    print(f"  Total links from file: {len(all_links)}", flush=True)
    
    # Check existing
    existing_urls = set()
    for row in c.execute("SELECT page_url FROM gov_raw WHERE site_name = ?", (SITE_NAME,)):
        existing_urls.add(row[0])
    
    new_links = [l for l in all_links if l not in existing_urls]
    print(f"  Existing: {len(existing_urls)}, New: {len(new_links)}", flush=True)
    
    if not new_links:
        print("  No new articles", flush=True)
        conn.close()
        return 0
    
    print(f"Fetching {len(new_links)} detail pages...", flush=True)
    count = 0
    for i, url in enumerate(new_links):
        detail = fetch_detail(url)
        
        if detail["error"]:
            print(f"  [{i+1}/{len(new_links)}] ERR: {url[-50:]} - {detail['error']}", flush=True)
            continue
        
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, publish_date, content) VALUES (?, ?, ?, ?, ?)",
            (SITE_NAME, url, detail["title"], detail["date"], detail["body"])
        )
        conn.commit()
        count += 1
        
        if (i+1) % 50 == 0:
            print(f"  [{i+1}/{len(new_links)}] Progress: {count} saved", flush=True)
        
        time.sleep(0.3)
    
    print(f"  Saved: {count}", flush=True)
    
    # FTS rebuild
    print("Rebuilding FTS...", flush=True)
    try:
        c.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
        conn.commit()
    except Exception as e:
        print(f"  FTS error: {e}", flush=True)
    
    conn.close()
    return count

if __name__ == "__main__":
    count = crawl()
    print(f"Done. {count} new articles.", flush=True)
