#!/usr/bin/env python3
"""
禄丰市人民政府 - 生态环境保护栏目爬虫
URL: https://www.ynlf.gov.cn/zfxxgk/fdzdgknr/wjjzdnhms/hms/sthj.htm
CMS: Visual SiteBuilder 9 (VSB9)
"""
import requests
import re
import sys
import os
import sqlite3
import json
import subprocess
from datetime import datetime
from bs4 import BeautifulSoup, NavigableString
from urllib.parse import urljoin

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9',
}
BASE_URL = 'https://www.ynlf.gov.cn'
LIST_URL = 'https://www.ynlf.gov.cn/zfxxgk/fdzdgknr/wjjzdnhms/hms/sthj.htm'

DB_PATH = '/mnt/data/search.db'


def get_conn():
    return sqlite3.connect(DB_PATH, timeout=60)


def extract_content_text(container):
    """Extract text with \n\n paragraph separation from a VSB9 content container.
    Handles nested <p>, <div>, <table> structures properly."""
    paragraphs = []
    tables_html = []
    
    for child in list(container.children):
        if child.name is None:
            continue  # text node
        elif child.name == 'table':
            tables_html.append(str(child))
            continue
        elif child.find('table'):
            # div wrapping a table - skip text, add table HTML
            for t in child.find_all('table', recursive=True):
                tables_html.append(str(t))
            # Skip text from this wrapper div (will be in table cells)
            continue
        
        # Extract text from this element
        if child.name == 'p':
            t = child.get_text(strip=True)
            if t:
                paragraphs.append(t)
        elif child.name in ['div', 'section', 'h1', 'h2', 'h3', 'h4', 'h5', 'h6']:
            t = child.get_text(strip=True)
            if t:
                paragraphs.append(t)
    
    return '\n\n'.join(paragraphs), tables_html


def extract_attachment_links(soup, page_url):
    """Extract image/attachment links from the page when content is minimal."""
    links = []
    for img in soup.select('img[src]'):
        src = urljoin(page_url, img['src'])
        alt = img.get('alt', '').strip()
        label = f"[图片: {alt}]" if alt else "[图片]"
        links.append(f"{label} {src}")
    for obj in soup.select('object[data]'):
        data = obj.get('data', '')
        if any(data.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx']):
            links.append(f"[附件PDF] {urljoin(page_url, data)}")
    for a in soup.select('a[href]'):
        href = a['href']
        full_url = urljoin(page_url, href)
        fname = a.get_text(strip=True) or href.split('/')[-1]
        if any(href.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.jpg', '.png', '.gif']):
            links.append(f"[附件: {fname}] {full_url}")
    return links


def clean_html(html, page_url=''):
    """Clean HTML and extract structured content with proper paragraph separation."""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Remove unwanted elements
    for tag in soup.select('script, style, iframe, .page_css, .label, hr, #div_vote_id'):
        tag.decompose()
    
    # Find main content container
    content = None
    for selector in ['div#vsb_content', 'div.content', 'div.v_news_content', 'div.vsb_content',
                     'div.TRS_Editor', 'div.xxgk_content', 'div.article-content']:
        el = soup.select_one(selector)
        if el:
            content = el
            break
    
    if not content:
        content = soup.body if soup.body else soup
    
    # Drill into v_news_content if it's nested inside #vsb_content
    vnc = content.select_one('div.v_news_content')
    if vnc:
        content = vnc
    
    # Extract text and tables
    text, tables_html = extract_content_text(content)
    
    # Check for attachments when content is very short
    links = extract_attachment_links(soup, page_url) if len(text) < 500 else []
    
    # Build final result
    result = text
    if tables_html:
        result += '\n\n' + '\n\n'.join(tables_html)
    if links:
        result += '\n\n' + '\n'.join(links)
    
    return result.strip()


def fetch_page(url):
    """Fetch a page with curl via subprocess."""
    try:
        result = subprocess.run([
            'curl', '-sS', '-L', '--max-time', '30', '-k',
            url
        ], capture_output=True, timeout=60)
        if result.returncode == 0 and result.stdout:
            text = result.stdout.decode('utf-8', errors='replace')
            return text
        print(f"  [WARN] curl exit={result.returncode} for {url}")
        return None
    except Exception as e:
        print(f"  [ERROR] fetch {url}: {e}")
        return None


def parse_list(html, base_url):
    """Parse VSB9 list page, return list of (title, url, date)."""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    
    for li in soup.select('li[id^="line_u6_"]'):
        a = li.select_one('a[href]')
        if not a:
            continue
        
        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        if not title or len(title) < 4:
            continue
        
        date_span = li.select_one('span.date')
        date = date_span.get_text(strip=True) if date_span else ''
        
        title = title.replace('\xa0', ' ').replace('\u00b7', '').strip()
        full_url = urljoin(base_url, href)
        
        items.append((title, full_url, date))
    
    return items


def get_next_page_url(current_url, soup):
    """Find next page URL for VSB9 pagination."""
    current_page = 1
    if current_url.endswith('sthj.htm'):
        current_page = 1
    else:
        m = re.search(r'/sthj/(\d+)\.htm', current_url)
        if m:
            current_page = int(m.group(1)) + 1
        else:
            next_a = soup.select_one('a[href*="sthj/"], span.p_next a')
            if next_a:
                return urljoin(current_url, next_a['href'])
            return None
    
    next_page = current_page + 1
    
    if next_page == 2:
        next_url = current_url.replace('sthj.htm', 'sthj/1.htm')
    else:
        next_url = re.sub(r'/sthj/\d+\.htm', f'/sthj/{next_page-1}.htm', current_url)
    
    return next_url


def sync_fts(conn, row_id):
    """Sync a single row to FTS table via subprocess (stdin to preserve Chinese chars)."""
    try:
        row = conn.execute('SELECT id, title, content, publish_date, source_url, site_name, summary FROM gov_raw WHERE id = ?', (row_id,)).fetchone()
        if not row:
            return
        rid, title, content, pub_date, src_url, site_name, summary = row
        summary_text = summary or (title or '')[:200]
        
        # Use stdin to pass SQL with proper Chinese characters
        sql = f"INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES({rid}, '{title.replace(chr(39), chr(39)*2)}', '{site_name.replace(chr(39), chr(39)*2)}', '{summary_text.replace(chr(39), chr(39)*2)}');\n"
        subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH], input=sql.encode('utf-8'), capture_output=True, timeout=10)
    except Exception as e:
        print(f"  [FTS ERROR] row_id={row_id}: {e}")


def crawl(pages=5):
    """Main crawl function."""
    all_items = []
    seen_urls = set()
    
    print(f"[START] 禄丰市-生态环境保护 爬虫 | pages={pages}")
    
    current_url = LIST_URL
    for pg in range(1, pages + 1):
        print(f"\n[LIST] Page {pg}: {current_url}")
        html = fetch_page(current_url)
        if not html:
            print(f"  [FAIL] Cannot fetch page {pg}")
            break
        
        items = parse_list(html, current_url)
        new_items = [it for it in items if it[1] not in seen_urls]
        for it in new_items:
            seen_urls.add(it[1])
        
        print(f"  Found {len(items)} items, {len(new_items)} new")
        all_items.extend(new_items)
        
        soup = BeautifulSoup(html, 'html.parser')
        next_url = get_next_page_url(current_url, soup)
        if not next_url:
            print(f"  No more pages")
            break
        current_url = next_url
    
    print(f"\n[TOTAL] {len(all_items)} items from {pg} pages")
    
    conn = get_conn()
    inserted = 0
    empty_content = 0
    
    for i, (title, url, date) in enumerate(all_items, 1):
        print(f"  [{i}/{len(all_items)}] {title[:40]}...", end=' ')
        
        existing = conn.execute('SELECT id FROM gov_raw WHERE source_url = ?', (url,)).fetchone()
        if existing:
            print(f"SKIP (exists)")
            continue
        
        html = fetch_page(url)
        if not html:
            print(f"FAIL (fetch)")
            continue
        
        soup = BeautifulSoup(html, 'html.parser')
        
        # Title
        content_title = title
        form = soup.select_one('form[name="_newscontent_fromname"]')
        if form:
            h1 = form.select_one('h1')
            if h1:
                t = h1.get_text(strip=True)
                if t and len(t) > 4:
                    content_title = t
        
        meta_title = soup.select_one('meta[name="ArticleTitle"]')
        if meta_title and meta_title.get('content'):
            mc = meta_title['content'].strip()
            if mc and len(mc) > 4:
                content_title = mc
        
        content_title = content_title.replace('\xa0', ' ').replace('\u00b7', '').strip()
        
        # Date
        content_date = date
        if form:
            date_text = form.get_text()
            m = re.search(r'日期[：:](\d{4})年(\d{1,2})月(\d{1,2})日', date_text)
            if m:
                content_date = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"
        
        meta_date = soup.select_one('meta[name="PubDate"]')
        if meta_date and meta_date.get('content'):
            md = meta_date['content'].strip()[:10]
            if re.match(r'\d{4}-\d{2}-\d{2}', md):
                content_date = md
        
        body = clean_html(html, url)
        
        if not body or len(body) < 10:
            empty_content += 1
        
        content_len = len(body) if body else 0
        site_name = '禄丰市人民政府'
        
        try:
            conn.execute(
                'INSERT INTO gov_raw (title, content, publish_date, source_url, site_name, summary, group_name, industry) VALUES (?,?,?,?,?,?,?,?)',
                (content_title, body or '', content_date, url, site_name, content_title[:200], '云南省楚雄州', '生态环境')
            )
            conn.commit()
            rid = conn.execute('SELECT last_insert_rowid()').fetchone()[0]
            sync_fts(conn, rid)
            inserted += 1
            print(f"OK ({content_len} chars)")
        except Exception as e:
            conn.rollback()
            print(f"DB ERROR: {e}")
    
    conn.close()
    
    print(f"\n[DONE] 总计 {len(all_items)} 条, 新增 {inserted} 条, 空正文 {empty_content} 条")


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--pages', type=int, default=5)
    args = parser.parse_args()
    crawl(args.pages)


if __name__ == '__main__':
    main()
