#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
遂溪县-生态环境局遂溪分局-政务公开 爬虫
Guangdong CMS, static HTML pagination: index_N.html
List: <li><a href="content/post_xxx.html">title</a><span.time>date</span>
Detail: <div class="content"> - tables preserved
"""

import re, time, os
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup, NavigableString

BASE_URL = "http://www.suixi.gov.cn/bmxxgk/sthjj/zwgk"
SCRIPT_DIR = os.path.dirname(os.path.abspath(__file__))

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml",
}

def safe_get(url, timeout=15):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=timeout)
            r.encoding = 'utf-8'
            return r
        except Exception as e:
            if attempt < 2:
                time.sleep(2)
            else:
                print(f"  [WARN] Failed to fetch {url}: {e}")
                return None

def strip_inline_tags(html_text):
    """Strip inline formatting tags (<b>, <span>, <font>, <strong>, <em>, <u>)
    while preserving text content and block-level structure.
    Also remove their style/class attributes from remaining tags."""
    # Remove inline tags completely (keep their inner text)
    for tag in ('b', 'strong', 'span', 'font', 'em', 'i', 'u', 's', 'sub', 'sup', 'small', 'mark'):
        html_text = re.sub(rf'</?{tag}[^>]*>', '', html_text, flags=re.I)
    # Remove style/class from remaining tags
    html_text = re.sub(r'\s+(style|class|lang|dir|align|valign)="[^"]*"', '', html_text)
    return html_text

def content_div_to_text(content_div, page_url):
    """Convert content div to structured text with tables preserved."""
    parts = []
    
    for child in content_div.children:
        name = getattr(child, 'name', None)
        
        if name == 'table':
            # Direct table
            parts.append(str(child))
            parts.append("\n\n")
        elif name == 'center':
            # Center with table inside
            inner_table = child.find('table')
            if inner_table:
                parts.append(str(inner_table))
            else:
                txt = child.get_text(' ', strip=True)
                if txt:
                    parts.append(txt)
            parts.append("\n\n")
        elif name in ('p', 'div'):
            # Check if this contains a table
            inner_table = child.find('table')
            if inner_table:
                # Split: text before table, table, text after table
                # For simplicity, just extract text from paragraphs and table from table
                txt_parts = []
                for sub in child.children:
                    if getattr(sub, 'name', None) == 'table':
                        parts.append(str(sub))
                        parts.append("\n\n")
                    elif getattr(sub, 'name', None) == 'center':
                        t = sub.find('table')
                        if t:
                            parts.append(str(t))
                        else:
                            ts = sub.get_text(' ', strip=True)
                            if ts:
                                txt_parts.append(ts)
                        parts.append("\n\n")
                    elif isinstance(sub, NavigableString):
                        t = str(sub).strip()
                        if t:
                            txt_parts.append(t)
                    else:
                        # Strip inline tags first
                        sub_html = str(sub)
                        sub_html = re.sub(
                            r'</?(?:span|b|strong|font|em|i|u|s|sub|sup|small|mark)[^>]*>',
                            '', sub_html, flags=re.I
                        )
                        clean_sub = BeautifulSoup(sub_html, 'html.parser')
                        t = clean_sub.get_text(' ', strip=True)
                        if t:
                            txt_parts.append(t)
                if txt_parts:
                    # Join text with double newline for paragraph separation
                    txt = ' '.join(txt_parts)
                    # Compress multiple spaces
                    txt = re.sub(r'\s+', ' ', txt).strip()
                    parts.append(txt)
                    parts.append("\n\n")
            else:
                # Plain paragraph - strip inline tags from HTML first to avoid
                # number splits like "202<strong>5</strong>" → "202 5"
                child_html = str(child)
                child_html = re.sub(
                    r'</?(?:span|b|strong|font|em|i|u|s|sub|sup|small|mark)[^>]*>',
                    '', child_html, flags=re.I
                )
                clean_soup = BeautifulSoup(child_html, 'html.parser')
                txt = clean_soup.get_text(' ', strip=True)
                if txt:
                    # Clean up inline formatting artifacts
                    txt = re.sub(r'\s+', ' ', txt).strip()
                    
                    # Check for images
                    imgs = child.find_all('img')
                    for img in imgs:
                        src = img.get('src', '')
                        if src:
                            parts.append(f"![image]({urljoin(page_url, src)})\n\n")
                    
                    parts.append(txt)
                    parts.append("\n\n")
        elif name == 'ol':
            for li in child.find_all('li', recursive=False):
                li_text = li.get_text(' ', strip=True)
                if li_text:
                    parts.append(f"  {li_text}")
            parts.append("\n\n")
        elif name == 'ul':
            for li in child.find_all('li', recursive=False):
                li_text = li.get_text(' ', strip=True)
                if li_text:
                    parts.append(f"- {li_text}")
            parts.append("\n\n")
        elif name == 'a':
            href = child.get('href', '')
            text = child.get_text(' ', strip=True)
            if "upload" in href.lower() or any(href.lower().endswith(ext) for ext in ('.doc', '.docx', '.pdf', '.xls', '.xlsx')):
                parts.append(f"[{text}]({urljoin(page_url, href)})\n\n")
            elif text and len(text) > 10:
                parts.append(text)
                parts.append("\n\n")
        elif name is None:
            t = str(child).strip()
            if t and len(t) > 5:
                parts.append(t)
                parts.append("\n\n")
        elif name == 'style':
            continue  # Skip style tags
    
    result = ''.join(parts).strip()
    result = re.sub(r'\n{4,}', '\n\n', result)
    # Clean up remaining multiple spaces
    result = re.sub(r' {3,}', '  ', result)
    return result

def parse_detail(url):
    r = safe_get(url)
    if not r:
        return None, None, None
    
    soup = BeautifulSoup(r.text, "html.parser")
    
    # Title
    title_tag = soup.find("title")
    title = title_tag.get_text(strip=True) if title_tag else ""
    
    # Content div
    content_div = soup.find("div", class_="content")
    if not content_div:
        for cls in ("content", "article", "main", "text", "Custom_Union", "TRS_Editor"):
            content_div = soup.find("div", class_=cls)
            if content_div:
                break
    if not content_div:
        content_div = soup.find("div", id=re.compile(r"content|article|zoom", re.I))
    
    content = ""
    if content_div:
        content = content_div_to_text(content_div, url)
    else:
        # Fallback: get all text
        content = soup.get_text('\n\n', strip=True)
    
    # Date
    date = ""
    for tag in soup.find_all(["span", "div", "p", "td"], class_=re.compile(r"(?:date|time|publish|source|info)", re.I)):
        m = re.search(r'(\d{4}[-/\.]\d{1,2}[-/\.]\d{1,2})', tag.get_text())
        if m:
            date = m.group(1).replace("/", "-").replace(".", "-")
            break
    
    if not date:
        for meta in soup.find_all("meta"):
            name = meta.get("name", "").lower()
            if name in ("publishdate", "pubdate", "date", "pubDate"):
                d = meta.get("content", "")
                m = re.search(r'(\d{4}[-/\.]\d{1,2}[-/\.]\d{1,2})', d)
                if m:
                    date = m.group(1).replace("/", "-").replace(".", "-")
                    break
    
    return title, content, date

def parse_list_page(url):
    r = safe_get(url)
    if not r:
        return []
    
    soup = BeautifulSoup(r.text, "html.parser")
    items = []
    
    for li in soup.find_all("li"):
        # Find article link (not category link with empty href)
        a = None
        for tag in li.find_all("a", href=True):
            h = tag["href"].strip()
            if h and "content/post_" in h:
                a = tag
                break
        if not a:
            continue
        
        # Get title from <a> text, but strip any date <span> inside
        # Some <a> contain <strong>title</strong><span.time>2026-05-11</span>
        title_html = str(a)
        # Remove <span> tags that contain dates
        title_html = re.sub(r'<span[^>]*class="?time"?[^>]*>.*?</span>', '', title_html, flags=re.I|re.DOTALL)
        title = BeautifulSoup(title_html, "html.parser").get_text(strip=True)
        
        href = a["href"].strip()
        
        if len(title) < 3:
            continue
        
        # Normalize URL
        if href.startswith("/"):
            href = urljoin(BASE_URL, href)
        elif not href.startswith("http"):
            href = urljoin(url, href)
        
        # Date
        date = ""
        time_span = li.find("span", class_="time")
        if time_span:
            date = time_span.get_text(strip=True)
        if not date:
            m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', li.get_text())
            if m:
                date = m.group(1)
        
        items.append((title, href, date))
    
    return items

def main():
    import sqlite3
    import sys as _SYS
    _MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
    if _MAX_PG is not None:
        print('[AutoPg] max_pages=' + str(_MAX_PG))
    # END AUTO PAGES
    
    # Find DB
    db_paths = [
        "/root/search.db",
        "/root/gov_crawler/search.db",
        os.path.join(SCRIPT_DIR, "search.db"),
    ]
    db_path = None
    for p in db_paths:
        if os.path.exists(p):
            db_path = p
            break
    if not db_path:
        db_path = "/root/search.db"
    
    print(f"=== 遂溪县-生态环境局遂溪分局-政务公开 ===")
    
    # Fetch all list pages
    all_items = []
    page_num = 1
    max_pages = 50
    
    while page_num <= max_pages:
        if _MAX_PG and page_num >= _MAX_PG: break
        if page_num == 1:
            url = BASE_URL + "/"
        else:
            url = f"{BASE_URL}/index_{page_num}.html"
        
        print(f"  Fetching page {page_num}: {url}")
        items = parse_list_page(url)
        if not items:
            print(f"  Empty page {page_num}, stopping")
            break
        
        print(f"    Found {len(items)} items")
        all_items.extend(items)
        page_num += 1
        time.sleep(0.3)
    
    print(f"\n  Total: {len(all_items)} items\n")
    
    if not all_items:
        print("No items found, exiting.")
        return
    
    # Save items list
    with open("/tmp/suixi_items.txt", "w") as f:
        for t, h, d in all_items:
            f.write(f"[{d}] {t}\n  {h}\n\n")
    
    # DB
    db = sqlite3.connect(db_path, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=OFF")
    
    db.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY,
            title TEXT,
            content TEXT,
            page_url TEXT UNIQUE,
            publish_date TEXT,
            site_name TEXT
        )
    """)
    # FTS 由 search.db 触发器 trg_gov_raw_fts_* 统一维护, 不再自建本地 gov_fts 表
    
    site_name = "遂溪县-生态环境局-政务公开"
    inserted = 0
    skipped = 0
    errors = 0
    
    for idx, (title, href, list_date) in enumerate(all_items):
        existing = db.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,)).fetchone()
        if existing:
            skipped += 1
            continue
        
        detail_title, content, detail_date = parse_detail(href)
        if not content:
            errors += 1
            if errors > 5:
                print("  Too many errors, stopping")
                break
            print(f"  [WARN] Empty content: {title[:50]} - skipping")
            continue
        
        final_title = detail_title or title
        final_date = detail_date or list_date
        
        content = content[:500000] if len(content) > 500000 else content
        
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (title, content, page_url, publish_date, site_name) VALUES (?, ?, ?, ?, ?)",
                (final_title, content, href, final_date, site_name)
            )
            rowid = db.execute("SELECT id FROM gov_raw WHERE page_url = ?", (href,)).fetchone()
            db.commit()
            inserted += 1
        except Exception as e:
            print(f"  [DB ERROR] {e}")
            db.rollback()
        
        if (idx + 1) % 10 == 0:
            print(f"  Progress: {idx+1}/{len(all_items)} (inserted={inserted}, skipped={skipped})")
            db.commit()
        
        time.sleep(0.3)
    
    db.commit()
    
    # Bulk FTS rebuild for this site

    
    db.close()
    
    print(f"\n========================================")
    print(f"Inserted: {inserted}/{len(all_items)} new")
    print(f"Skipped (already exist): {skipped}")
    print(f"Errors: {errors}")

if __name__ == "__main__":
    main()
