#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
犍为县人民政府 - 重大民生信息→环境保护 爬虫
URL: http://www.qianwei.gov.cn/qwx/hjbht/jcxzdly.shtml
CMS: 开普云/UCAP (UCAPTITLE, UCAPCONTENT, PUBLISHTIME)
列表: <ul class="list_ul_a"> → <li><a href="..." title="TITLE">TITLE</a><span>DATE</span></li>
分页: jcxzdly.shtml(第1页) / jcxzdly_{N}.shtml(第2页起) 21条/页 34页共714条
详情: /qwx/hjbht/{YYYYMM}/{uuid}.shtml
详情结构: h3>UCAPTITLE(标题), PUBLISHTIME(日期), div.main#new_content>UCAPCONTENT(正文)
"""
import re
import sys
import json
import subprocess
from urllib.parse import urljoin

from bs4 import BeautifulSoup

# ── 配置 ──
BASE_URL = "http://www.qianwei.gov.cn"
SITE_NAME = "犍为县人民政府-环境保护"
GROUP = "四川省乐山市"
INDUSTRY = "生态环境"
DB_PATH = "/mnt/data/search.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def clean_text(text):
    if not text:
        return ""
    text = re.sub(r'(&middot;|&nbsp;|\s)+', ' ', text)
    return text.strip()


def clean_content_html(html_content, detail_url):
    """提取UCAP正文，表格保留HTML，附件嵌入"""
    if not html_content:
        return ""
    soup = BeautifulSoup(html_content, 'html.parser')

    # Find content - UCAPCONTENT tag
    ucap = soup.find('ucapcontent')
    content = ucap if ucap else soup.select_one('div.main#new_content')
    if not content:
        content = soup.select_one('div.detail')

    if not content:
        return ""

    # Remove unwanted elements
    for tag in content.select('script, style, link, #asbox, .zjsc'):
        tag.decompose()

    tables_html = []
    paragraphs = []

    for child in list(content.children):
        if child.name is None:
            # Text node (handles <br> separated content)
            t = str(child).strip()
            if t and len(t) > 2:
                # Also check for <br> in raw HTML
                paragraphs.append(t)
            continue
        if child.name in ['br']:
            continue
        if child.name == 'table':
            tables_html.append(str(child))
            continue
        if child.find('table'):
            for t in child.find_all('table', recursive=True):
                tables_html.append(str(t))
            continue

        if child.name in ['p', 'div', 'section', 'h1', 'h2', 'h3', 'h4']:
            t = child.get_text(strip=True)
            if t:
                paragraphs.append(t)

    text = '\n\n'.join(paragraphs)

    # Attachments/images when content short
    links = []
    if len(text) < 500:
        for img in content.select('img[src]'):
            src = urljoin(detail_url, img['src'])
            alt = img.get('alt', '').strip()
            label = f"[图片: {alt}]" if alt else "[图片]"
            links.append(f"{label} {src}")
        for a in content.select('a[href]'):
            href = a['href']
            full_url = urljoin(detail_url, href)
            fname = a.get_text(strip=True) or href.split('/')[-1]
            if any(href.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx']):
                links.append(f"[附件: {fname}] {full_url}")

    result = text
    if tables_html:
        result += '\n\n' + '\n\n'.join(tables_html)
    if links:
        result += '\n\n' + '\n'.join(links)

    return result.strip()


def fetch_page(url):
    """Fetch page with curl."""
    try:
        result = subprocess.run([
            'curl', '-sS', '-L', '--max-time', '15', '-k',
            url
        ], capture_output=True, timeout=30)
        if result.returncode == 0 and result.stdout:
            return result.stdout.decode('utf-8', errors='replace')
        print(f"  [WARN] curl exit={result.returncode} for {url}")
        return None
    except Exception as e:
        print(f"  [ERROR] {e}")
        return None


def parse_list(html, page_url):
    """Parse list page, return list of (title, url, date)."""
    items = []
    soup = BeautifulSoup(html, 'html.parser')

    for li in soup.select('ul.list_ul_a li'):
        a = li.select_one('a[href]')
        span = li.select_one('span')
        if not a:
            continue

        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        if not title or len(title) < 4:
            continue

        date = span.get_text(strip=True) if span else ''

        title = clean_text(title)
        full_url = urljoin(BASE_URL, href)

        items.append((title, full_url, date))

    return items


def get_list_url(page_num):
    """Get list page URL for given page number."""
    if page_num == 1:
        return f"{BASE_URL}/qwx/hjbht/jcxzdly.shtml"
    return f"{BASE_URL}/qwx/hjbht/jcxzdly_{page_num}.shtml"


def sync_fts(row_id, title, site_name, summary):
    """Sync FTS via subprocess stdin."""
    try:
        title_esc = title.replace("'", "''")
        site_esc = site_name.replace("'", "''")
        summ_esc = summary[:200].replace("'", "''")
        sql = f"INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES({row_id}, '{title_esc}', '{site_esc}', '{summ_esc}');\n"
        subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH], input=sql.encode('utf-8'), capture_output=True, timeout=10)
    except Exception as e:
        print(f"  [FTS ERROR] row_id={row_id}: {e}")


def crawl(pages=5):
    all_items = []
    seen_urls = set()

    print(f"[START] {SITE_NAME} | pages={pages}")

    for pg in range(1, pages + 1):
        url = get_list_url(pg)
        print(f"\n[LIST] Page {pg}: {url}")
        html = fetch_page(url)
        if not html:
            print(f"  [FAIL] Cannot fetch page {pg}")
            break

        items = parse_list(html, url)
        new_items = [it for it in items if it[1] not in seen_urls]
        for it in new_items:
            seen_urls.add(it[1])

        print(f"  Found {len(items)} items, {len(new_items)} new")
        all_items.extend(new_items)

    print(f"\n[TOTAL] {len(all_items)} items from {pages} pages")

    conn = __import__('sqlite3').connect(DB_PATH)
    inserted = 0
    empty_content = 0
    total_chars = 0

    for i, (title, url, date) in enumerate(all_items, 1):
        print(f"  [{i}/{len(all_items)}] {title[:40]}...", end=' ')

        existing = conn.execute('SELECT id FROM gov_raw WHERE source_url = ?', (url,)).fetchone()
        if existing:
            print(f"SKIP (exists)")
            continue

        html = fetch_page(url)
        if not html:
            print(f"FAIL (fetch)")
            continue

        soup = BeautifulSoup(html, 'html.parser')

        # Title from UCAPTITLE
        content_title = title
        ucap_title = soup.find('ucaptitle')
        if ucap_title:
            t = ucap_title.get_text(strip=True)
            if t and len(t) > 4:
                content_title = t

        content_title = clean_text(content_title)

        # Date from PUBLISHTIME
        content_date = date
        pub_time = soup.find('publishtime')
        if pub_time:
            m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', pub_time.get_text())
            if m:
                content_date = m.group(1)

        # Body
        body = clean_content_html(html, url)

        if not body or len(body) < 10:
            empty_content += 1

        content_len = len(body) if body else 0
        total_chars += content_len

        try:
            conn.execute(
                'INSERT INTO gov_raw (title, content, publish_date, source_url, site_name, summary, group_name, industry) VALUES (?,?,?,?,?,?,?,?)',
                (content_title, body or '', content_date, url, SITE_NAME, content_title[:200], GROUP, INDUSTRY)
            )
            conn.commit()
            rid = conn.execute('SELECT last_insert_rowid()').fetchone()[0]
            sync_fts(rid, content_title, SITE_NAME, content_title[:200])
            inserted += 1
            print(f"OK ({content_len} chars)")
        except Exception as e:
            conn.rollback()
            print(f"DB ERROR: {e}")

    conn.close()
    print(f"\n[DONE] 总计 {len(all_items)} 条, 新增 {inserted} 条, 空正文 {empty_content} 条, 总字数 {total_chars}")


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--pages', type=int, default=5)
    args = parser.parse_args()
    crawl(args.pages)


if __name__ == '__main__':
    main()
