#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
广德市人民政府 - 基层政务公开→市生态环境分局→批准结果信息 爬虫
URL: https://www.guangde.gov.cn/Jczwgk/showList/548/102002000/page_1.html
CMS: 广德自定义 基层政务公开平台
列表: <ul class="m-liststyle2"> → <li><span>DATE</span><a href="..." title="TITLE"></a></li>
分页: /page_{n}.html (15条/页, 28页共416条)
详情: /Jczwgk/show/{id}.html
详情结构: h1.u-lgtit(标题), span发布时间:, div.m-dttexts.j-fontContent#zoom(正文)
"""
import re
import sys
import json
import subprocess
from urllib.parse import urljoin

from bs4 import BeautifulSoup

# ── 配置 ──
BASE_URL = "https://www.guangde.gov.cn"
LIST_URL_TMPL = BASE_URL + "/Jczwgk/showList/548/102002000/page_{n}.html"
SITE_NAME = "广德市人民政府-环评批准结果"
GROUP = "安徽"
DB_PATH = "/mnt/data/search.db"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def clean_text(text):
    if not text:
        return ""
    text = re.sub(r'(&middot;|&nbsp;|\s)+', ' ', text)
    return text.strip()


def clean_content_html(html_content, detail_url):
    """提取正文，表格保留HTML，附件嵌入"""
    if not html_content:
        return ""
    soup = BeautifulSoup(html_content, 'html.parser')

    # 找内容容器
    content = None
    for sel in ['div.m-dttexts.j-fontContent', 'div.j-fontContent', 'div#zoom',
                'div.g-detailbox', 'div.m-dttexts']:
        c = soup.select_one(sel)
        if c:
            content = c
            break

    if not content:
        return ""

    # 移除无用元素
    for tag in content.select('script, style, link'):
        tag.decompose()

    tables_html = []
    paragraphs = []

    # 遍历直接子元素
    for child in list(content.children):
        if child.name is None:
            continue
        if child.name == 'table':
            tables_html.append(str(child))
            continue
        if child.find('table'):
            # div包裹的表格，跳过文本但保留表格HTML
            for t in child.find_all('table', recursive=True):
                tables_html.append(str(t))
            continue

        # 提取p/div/text段落
        if child.name in ['p', 'div', 'section', 'h1', 'h2', 'h3', 'h4']:
            t = child.get_text(strip=True)
            if t:
                paragraphs.append(t)

    text = '\n\n'.join(paragraphs)

    # 内容很短时检查附件/图片
    links = []
    if len(text) < 500:
        for img in content.select('img[src]'):
            src = urljoin(detail_url, img['src'])
            alt = img.get('alt', '').strip()
            label = f"[图片: {alt}]" if alt else "[图片]"
            links.append(f"{label} {src}")
        for a in content.select('a[href]'):
            href = a['href']
            full_url = urljoin(detail_url, href)
            fname = a.get_text(strip=True) or href.split('/')[-1]
            if any(href.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx']):
                links.append(f"[附件: {fname}] {full_url}")

    result = text
    if tables_html:
        result += '\n\n' + '\n\n'.join(tables_html)
    if links:
        result += '\n\n' + '\n'.join(links)

    return result.strip()


def fetch_page(url):
    """Fetch page with curl via subprocess."""
    try:
        result = subprocess.run([
            'curl', '-sS', '-L', '--max-time', '30', '-k',
            url
        ], capture_output=True, timeout=60)
        if result.returncode == 0 and result.stdout:
            return result.stdout.decode('utf-8', errors='replace')
        print(f"  [WARN] curl exit={result.returncode} for {url}")
        return None
    except Exception as e:
        print(f"  [ERROR] {e}")
        return None


def parse_list(html, page_url):
    """Parse list page, return list of (title, url, date)."""
    items = []
    soup = BeautifulSoup(html, 'html.parser')

    for li in soup.select('div.m-liststyle2 ul li'):
        a = li.select_one('a[href]')
        span = li.select_one('span')
        if not a:
            continue

        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        if not title or len(title) < 4:
            continue

        date = span.get_text(strip=True) if span else ''

        title = clean_text(title)
        full_url = urljoin(BASE_URL, href)

        items.append((title, full_url, date))

    return items


def get_next_page_url(current_url, page_num):
    """Get next page URL based on page number."""
    # page_1.html → page_2.html
    next_url = re.sub(r'page_\d+\.html', f'page_{page_num + 1}.html', current_url)
    return next_url


def sync_fts(row_id, title, site_name, summary):
    """Sync FTS via subprocess stdin (preserves Chinese chars)."""
    try:
        title_esc = title.replace("'", "''")
        site_esc = site_name.replace("'", "''")
        summ_esc = summary[:200].replace("'", "''")
        sql = f"INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES({row_id}, '{title_esc}', '{site_esc}', '{summ_esc}');\n"
        subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH], input=sql.encode('utf-8'), capture_output=True, timeout=10)
    except Exception as e:
        print(f"  [FTS ERROR] row_id={row_id}: {e}")


def crawl(pages=5):
    all_items = []
    seen_urls = set()

    print(f"[START] {SITE_NAME} | pages={pages}")

    for pg in range(1, pages + 1):
        url = LIST_URL_TMPL.format(n=pg)
        print(f"\n[LIST] Page {pg}: {url}")
        html = fetch_page(url)
        if not html:
            print(f"  [FAIL] Cannot fetch page {pg}")
            break

        items = parse_list(html, url)
        new_items = [it for it in items if it[1] not in seen_urls]
        for it in new_items:
            seen_urls.add(it[1])

        print(f"  Found {len(items)} items, {len(new_items)} new")
        all_items.extend(new_items)

    print(f"\n[TOTAL] {len(all_items)} items from {pages} pages")

    conn = __import__('sqlite3').connect(DB_PATH)
    inserted = 0
    empty_content = 0
    total_chars = 0

    for i, (title, url, date) in enumerate(all_items, 1):
        print(f"  [{i}/{len(all_items)}] {title[:40]}...", end=' ')

        existing = conn.execute('SELECT id FROM gov_raw WHERE source_url = ?', (url,)).fetchone()
        if existing:
            print(f"SKIP (exists)")
            continue

        html = fetch_page(url)
        if not html:
            print(f"FAIL (fetch)")
            continue

        soup = BeautifulSoup(html, 'html.parser')

        # Title from detail page
        h1 = soup.select_one('h1.u-lgtit')
        content_title = h1.get_text(strip=True) if h1 else title
        content_title = clean_text(content_title)

        # Date from detail page
        content_date = date
        date_el = soup.select_one('span:contains("发布时间")')
        if date_el:
            m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', date_el.get_text())
            if m:
                content_date = m.group(1)

        # Body
        body = clean_content_html(html, url)

        if not body or len(body) < 10:
            empty_content += 1

        content_len = len(body) if body else 0
        total_chars += content_len

        try:
            conn.execute(
                'INSERT INTO gov_raw (title, content, publish_date, source_url, site_name, summary, group_name, industry) VALUES (?,?,?,?,?,?,?,?)',
                (content_title, body or '', content_date, url, SITE_NAME, content_title[:200], GROUP, '环评公示')
            )
            conn.commit()
            rid = conn.execute('SELECT last_insert_rowid()').fetchone()[0]
            sync_fts(rid, content_title, SITE_NAME, content_title[:200])
            inserted += 1
            print(f"OK ({content_len} chars)")
        except Exception as e:
            conn.rollback()
            print(f"DB ERROR: {e}")

    conn.close()
    print(f"\n[DONE] 总计 {len(all_items)} 条, 新增 {inserted} 条, 空正文 {empty_content} 条, 总字数 {total_chars}")


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--pages', type=int, default=5)
    args = parser.parse_args()
    crawl(args.pages)


if __name__ == '__main__':
    main()
