#!/usr/bin/env python3
"""
沅江市-建设项目环评
http://www.yuanjiang.gov.cn/20839/index.htm
CMS: 自有CMS (GB2312)
分页: index_N.htm (N从0开始，首页index.htm，第2页index_1.htm...第10页index_9.htm)
详情: content_{id}.html
"""
import requests
import sqlite3
import re
import os
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = "/root/gov_crawler/search.db"
SITE_NAME = "沅江市-建设项目环评"
LIST_URL = "http://www.yuanjiang.gov.cn/20839/index.htm"
BASE_URL = "http://www.yuanjiang.gov.cn/20839/"
TABLE_NAME = "gov_raw"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

def fetch_page(page):
    """Get list page content for the given page number (0-indexed)"""
    if page == 0:
        url = LIST_URL
    else:
        url = f"{BASE_URL}index_{page}.htm"
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "gb2312"
    return r.text

def extract_items(html):
    """Extract article links and dates from list page"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    
    # Find the lists div
    lists_div = soup.find("div", class_="lists")
    if not lists_div:
        lists_div = soup
    
    # Look for li items with a and span (date)
    for li in lists_div.find_all("li"):
        a_tag = li.find("a")
        if not a_tag or not a_tag.get("href"):
            continue
        href = a_tag["href"]
        title = a_tag.get_text(strip=True)
        if not title or len(title) < 10:
            continue
        if "content_" not in href:
            continue
        
        # Build full URL
        if not href.startswith("http"):
            if href.startswith("/"):
                href = "http://www.yuanjiang.gov.cn" + href
            elif href.startswith("."):
                # Relative to list URL
                href = f"http://www.yuanjiang.gov.cn/20839/{href}"
            else:
                href = BASE_URL + href
        
        # Get date from span
        date_str = ""
        span = li.find("span")
        if span:
            date_str = span.get_text(strip=True)
        
        # Clean title (remove &nbsp; etc)
        title = re.sub(r"\s*\u00a0+\s*", " ", title)
        title = re.sub(r"\s{2,}", " ", title).strip()
        
        items.append({"title": title, "url": href, "date": date_str})
    
    return items

def parse_pagination(html):
    """Extract total count and max page from pagination div"""
    total = 0
    max_page = 1
    current = 1
    
    # Find fenye div
    fenye_match = re.search(r'共(\d+)条.*?(\d+)/(\d+)页', html)
    if fenye_match:
        total = int(fenye_match.group(1))
        current = int(fenye_match.group(2))
        max_page = int(fenye_match.group(3))
        print(f"  Pagination: {total}条, {current}/{max_page}页", flush=True)
    
    return total, max_page

def fetch_detail(url):
    """Fetch detail page content"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "gb2312"
        soup = BeautifulSoup(r.text, "html.parser")
        
        for tag in soup(["script", "style", "nav", "footer", "header", "aside"]):
            tag.decompose()
        
        content_div = (
            soup.find("div", id="zoom")
            or soup.find("div", class_="yea-con")
            or soup.find("div", class_="article-con")
            or soup.find("div", id="content")
        )
        
        if content_div:
            content = content_div.get_text("\n", strip=True)
        else:
            content = soup.get_text("\n", strip=True)
        
        content = re.sub(r"\n{3,}", "\n\n", content)
        content = re.sub(r" {2,}", " ", content)
        
        return content.strip()
    except Exception as e:
        print(f"  [ERROR] detail: {e}", flush=True)
        return ""

def main():
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    
    # Check existing URLs
    existing = set()
    for row in c.execute("SELECT url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        existing.add(row[0])
    
    latest_db_date = "1900-01-01"
    row = c.execute("SELECT MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()
    if row and row[0]:
        latest_db_date = row[0]
    conn.close()
    
    print(f"Latest DB date: {latest_db_date}", flush=True)
    print(f"Existing URLs: {len(existing)}", flush=True)
    
    # First, get page 1 to find total pages
    html = fetch_page(0)
    total_count, max_page = parse_pagination(html)
    print(f"Total pages: {max_page}, Expected total: {total_count}", flush=True)
    
    # Scrape all pages
    all_items = []
    for page in range(max_page):
        if page > 0:
            html = fetch_page(page)
        items = extract_items(html)
        print(f"  Page {page+1}/{max_page}: {len(items)} items", flush=True)
        all_items.extend(items)
    
    print(f"Total list items: {len(all_items)}", flush=True)
    
    # Fetch details for new items
    new_count = 0
    skip_count = 0
    error_count = 0
    
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    
    for item in all_items:
        url = item["url"]
        title = item["title"]
        date_str = item["date"]
        
        if url in existing:
            skip_count += 1
            continue
        
        content = fetch_detail(url)
        if not content or len(content) < 100:
            print(f"  [SHORT] {title[:30]}... ({len(content)})", flush=True)
            if not content:
                error_count += 1
                continue
        
        if not date_str:
            try:
                r2 = requests.get(url, headers=HEADERS, timeout=15)
                r2.encoding = "gb2312"
                date_m = re.search(r"(\d{4}[-/]\d{1,2}[-/]\d{1,2})", r2.text)
                if date_m:
                    date_str = date_m.group(1).replace("/", "-")
            except:
                pass
        
        if not date_str:
            date_str = datetime.now().strftime("%Y-%m-%d")
        
        date_str = re.sub(r"[^\d-]", "", date_str)[:10]
        
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (title, content, url, site_name, publish_date)
            VALUES (?, ?, ?, ?, ?)
        """, (title, content, url, SITE_NAME, date_str))
        new_count += 1
        
        if new_count % 20 == 0:
            conn.commit()
            print(f"  Progress: {new_count} new / {skip_count} skip", flush=True)
    
    conn.commit()
    conn.close()
    
    print(f"\n{'='*50}", flush=True)
    print(f"Site: {SITE_NAME}", flush=True)
    print(f"Total list items: {len(all_items)}", flush=True)
    print(f"New: {new_count}", flush=True)
    print(f"Skipped (existing): {skip_count}", flush=True)
    print(f"Errors: {error_count}", flush=True)
    print(f"{'='*50}", flush=True)

if __name__ == "__main__":
    main()
