#!/usr/bin/env python3
"""
沅江市-建设项目环评
http://www.yuanjiang.gov.cn/20839/index.htm
CMS: 自有CMS (GB2312)
分页: index_N.htm (N从0开始，首页index.htm，第2页index_1.htm...第10页index_9.htm)
详情: content_{id}.html
"""
import requests

def _safe_summary(text, max_len=500):
    """Truncate text safely without breaking HTML tags"""
    if len(text) <= max_len:
        return text
    # Truncate, then cut at last complete tag boundary
    truncated = text[:max_len]
    # Count unclosed < tags
    opens = truncated.count('<')
    closes = truncated.count('</')
    # If < > pair count mismatch, might be cut mid-tag - rsplit at last safe <
    if '<' in truncated:
        last_open = truncated.rfind('<')
        last_close = truncated.rfind('>')
        if last_open > last_close:
            # Cut in middle of an opening tag
            truncated = truncated[:last_open]
        elif truncated.count('<') % 2 != 0:
            # Odd number of '<' - might be a self-closing tag or cut mid-entity
            pass
    return truncated

import sqlite3
import re
import os
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
SITE_NAME = "沅江市-建设项目环评"
LIST_URL = "http://www.yuanjiang.gov.cn/20839/index.htm"
BASE_URL = "http://www.yuanjiang.gov.cn/20839/"
TABLE_NAME = "gov_raw"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}

def fetch_page(page):
    """Get list page content for the given page number (0-indexed)"""
    if page == 0:
        url = LIST_URL
    else:
        url = f"{BASE_URL}index_{page}.htm"
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "gb2312"
    return r.text

def extract_items(html):
    """Extract article links and dates from list page"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    
    # Find the lists div
    lists_div = soup.find("div", class_="lists")
    if not lists_div:
        lists_div = soup
    
    # Look for li items with a and span (date)
    for li in lists_div.find_all("li"):
        a_tag = li.find("a")
        if not a_tag or not a_tag.get("href"):
            continue
        href = a_tag["href"]
        title = a_tag.get_text(strip=True)
        if not title or len(title) < 10:
            continue
        if "content_" not in href:
            continue
        
        # Build full URL
        if not href.startswith("http"):
            if href.startswith("/"):
                href = "http://www.yuanjiang.gov.cn" + href
            elif href.startswith("."):
                # Relative to list URL
                href = f"http://www.yuanjiang.gov.cn/20839/{href}"
            else:
                href = BASE_URL + href
        
        # Get date from span
        date_str = ""
        span = li.find("span")
        if span:
            date_str = span.get_text(strip=True)
        
        # Clean title (remove &nbsp; etc)
        title = re.sub(r"\s*\u00a0+\s*", " ", title)
        title = re.sub(r"\s{2,}", " ", title).strip()
        
        items.append({"title": title, "url": href, "date": date_str})
    
    return items

def parse_pagination(html):
    """Extract total count and max page from pagination div"""
    total = 0
    max_page = 1
    current = 1
    
    # Find fenye div
    fenye_match = re.search(r'共(\d+)条.*?(\d+)/(\d+)页', html)
    if fenye_match:
        total = int(fenye_match.group(1))
        current = int(fenye_match.group(2))
        max_page = int(fenye_match.group(3))
        print(f"  Pagination: {total}条, {current}/{max_page}页", flush=True)
    
    return total, max_page

def _extract_clean_text(element):
    """Extract content: plain text paragraphs + HTML tables preserved.
    Uses regex-based text extraction (not BS4 get_text) to preserve paragraph breaks.
    """
    html_str = str(element)
    html_str = re.sub(r'<span[^>]*>', '', html_str, flags=re.IGNORECASE)
    html_str = re.sub(r'</span>', '', html_str, flags=re.IGNORECASE)
    html_str = re.sub(r'<font[^>]*>', '', html_str, flags=re.IGNORECASE)
    html_str = re.sub(r'</font>', '', html_str, flags=re.IGNORECASE)

    soup = BeautifulSoup(html_str, 'html.parser')

    tables = []
    for table in soup.find_all('table'):
        idx = len(tables)
        tables.append(str(table))
        table.replace_with(f'__TABLE_{idx}__')

    clean_tables = []
    for t in tables:
        for attr in ['style', 'class', 'width', 'height', 'valign', 'align', 'border', 'cellpadding', 'cellspacing']:
            t = re.sub(rf' {attr}="[^"]*"', '', t)
        t = re.sub(r'<p>\s*', '', t)
        t = re.sub(r'\s*</p>', '', t)
        t = re.sub(r'<span[^>]*>', '', t)
        t = re.sub(r'</span>', '', t)
        t = re.sub(r'<font[^>]*>', '', t)
        t = re.sub(r'</font>', '', t)
        clean_tables.append(t)

    text = str(soup)

    replacement_nl2 = chr(10) + chr(10)
    replacement_nl = chr(10)
    text = re.sub(r'</p>\s*', replacement_nl2, text, flags=re.IGNORECASE)
    text = re.sub(r'</div>\s*', replacement_nl2, text, flags=re.IGNORECASE)
    text = re.sub(r'<br\s*/?>\s*', replacement_nl, text, flags=re.IGNORECASE)

    text = re.sub(r'<[^>]+>', '', text)

    import html as html_mod
    text = html_mod.unescape(text)

    text = re.sub(r'\n{3,}', chr(10)*2, text)
    text = re.sub(r'[ 	    ]+', ' ', text)
    text = re.sub(chr(10) + ' ', chr(10), text)
    text = re.sub(' ' + chr(10), chr(10), text)
    text = text.strip()

    for idx, clean_table in enumerate(clean_tables):
        text = text.replace(f'__TABLE_{idx}__', chr(10)*2 + clean_table + chr(10)*2)

    text = re.sub(r'\n{3,}', chr(10)*2, text)
    text = re.sub(r'[ 	    ]+', ' ', text)
    return text.strip()

def fetch_detail(url):
    """Fetch detail page content"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "gb2312"
        soup = BeautifulSoup(r.text, "html.parser")
        
        for tag in soup(["script", "style", "nav", "footer", "header", "aside"]):
            tag.decompose()
        
        content_div = (
            soup.find("div", id="zoom")
            or soup.find("div", class_="yea-con")
            or soup.find("div", class_="article-con")
            or soup.find("div", id="content")
        )
        
        content = _extract_clean_text(content_div if content_div else soup)
        
        content = re.sub("\n{3,}", "\n\n", content)
        content = re.sub(r" {2,}", " ", content)
        
        return content.strip()
    except Exception as e:
        print(f"  [ERROR] detail: {e}", flush=True)
        return ""

def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    # Check existing URLs
    existing = set()
    for row in c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        existing.add(row[0])
    
    latest_db_date = "1900-01-01"
    row = c.execute("SELECT MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()
    if row and row[0]:
        latest_db_date = row[0]
    conn.close()
    
    print(f"Latest DB date: {latest_db_date}", flush=True)
    print(f"Existing URLs: {len(existing)}", flush=True)
    
    # First, get page 1 to find total pages
    html = fetch_page(0)
    total_count, max_page = parse_pagination(html)
    print(f"Total pages: {max_page}, Expected total: {total_count}", flush=True)
    
    # Scrape all pages
    all_items = []
    for page in range(max_page):
        if page > 0:
            html = fetch_page(page)
        items = extract_items(html)
        print(f"  Page {page+1}/{max_page}: {len(items)} items", flush=True)
        all_items.extend(items)
    
    print(f"Total list items: {len(all_items)}", flush=True)
    
    # Fetch details for new items
    new_count = 0
    skip_count = 0
    error_count = 0
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    for item in all_items:
        url = item["url"]
        title = item["title"]
        date_str = item["date"]
        
        if url in existing:
            skip_count += 1
            continue
        
        content = fetch_detail(url)
        if not content or len(content) < 100:
            print(f"  [SHORT] {title[:30]}... ({len(content)})", flush=True)
            if not content:
                error_count += 1
                continue
        
        if not date_str:
            try:
                r2 = requests.get(url, headers=HEADERS, timeout=15)
                r2.encoding = "gb2312"
                date_m = re.search(r"(\d{4}[-/]\d{1,2}[-/]\d{1,2})", r2.text)
                if date_m:
                    date_str = date_m.group(1).replace("/", "-")
            except:
                pass
        
        if not date_str:
            date_str = datetime.now().strftime("%Y-%m-%d")
        
        date_str = re.sub(r"[^\d-]", "", date_str)[:10]
        
        c.execute("""
            INSERT OR IGNORE INTO gov_raw (site_name, source_url, page_url, title, publish_date, content, summary, date_rank)
            VALUES (?, ?, ?, ?, ?, ?, ?, ?)
        """, (SITE_NAME, BASE_URL.strip('/'), url, title, date_str, content, _safe_summary(content), int(date_str.replace("-", ""))))
        new_count += 1
        
        if new_count % 20 == 0:
            conn.commit()
            print(f"  Progress: {new_count} new / {skip_count} skip", flush=True)
    
    conn.commit()
    conn.close()
    
    print(f"\n{'='*50}", flush=True)
    print(f"Site: {SITE_NAME}", flush=True)
    print(f"Total list items: {len(all_items)}", flush=True)
    print(f"New: {new_count}", flush=True)
    print(f"Skipped (existing): {skip_count}", flush=True)
    print(f"Errors: {error_count}", flush=True)
    print(f"{'='*50}", flush=True)

if __name__ == "__main__":
    main()
