#!/usr/bin/env python3
import os
"""Crawler for 淮南市大通区人民政府 - 通知公告 (FIXED: UTF-8 encoding)"""
import sys, os, re, json, time, subprocess
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))

import requests
from bs4 import BeautifulSoup

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES


SITE_NAME = '淮南市大通区人民政府-通知公告'
DOMAIN = 'www.hndt.gov.cn'
BASE = 'https://www.hndt.gov.cn'
COLUMN_ID = '6787542'
MIN_PG = _MAX_PG if _MAX_PG else 42

HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def http_get(url):
    """GET with forced UTF-8 decoding — server omits charset in Content-Type, so requests auto-detects wrong encoding"""
    r = requests.get(url, headers=HEADERS, timeout=10)
    r.encoding = 'utf-8'  # Force UTF-8 — the meta charset says utf-8 but HTTP header omits it
    return r.text


def extract_list(html):
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    ul = soup.find('ul', class_=re.compile(r'doc_list'))
    if not ul:
        return items
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '')
        if not href:
            continue
        title = a.get('title', '') or a.get_text(strip=True)
        date_span = li.find('span', class_='date')
        date = date_span.get_text(strip=True) if date_span else ''
        if href.startswith('/'):
            href = BASE + href
        elif not href.startswith('http'):
            href = BASE + '/' + href.lstrip('/')
        items.append({'url': href, 'title': title, 'date': date})
    return items


def parse_detail(html):
    soup = BeautifulSoup(html, 'html.parser')
    title = ''
    date = ''
    content = ''
    attachments = []

    h1 = soup.find('h1', class_='newstitle')
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        meta = soup.find('meta', attrs={'name': 'ArticleTitle'})
        if meta:
            title = meta.get('content', '')

    date_span = soup.find('span', class_='sp')
    if date_span:
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', date_span.get_text())
        if m:
            date = m.group(1)
    if not date:
        meta = soup.find('meta', attrs={'name': 'PubDate'})
        if meta:
            pub = meta.get('content', '')
            m = re.search(r'(\d{4}-\d{2}-\d{2})', pub)
            if m:
                date = m.group(1)

    content_div = soup.find('div', class_=re.compile(r'wzcon'))
    if not content_div:
        content_div = soup.find('div', class_=re.compile(r'j-fontContent'))
    if content_div:
        for a_tag in content_div.find_all('a'):
            ahref = a_tag.get('href', '')
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|zip|rar|txt)(\?|$)', ahref, re.I):
                atext = a_tag.get_text(strip=True) or ahref.split('/')[-1]
                if not ahref.startswith('http'):
                    ahref = BASE + '/' + ahref.lstrip('/')
                a_tag.replace_with(f'[{atext}]({ahref})')
                attachments.append({'title': atext, 'url': ahref})

        texts = []
        for child in content_div.children:
            if hasattr(child, 'name') and child.name == 'table':
                texts.append('\n' + str(child) + '\n')
            elif hasattr(child, 'get_text'):
                txt = body_text(child)
                txt = re.sub(r'\s+', ' ', txt).strip()
                if txt:
                    texts.append(txt)
            elif isinstance(child, str):
                txt = child.strip()
                if txt:
                    texts.append(txt)
        content = '\n\n'.join(texts)
        content = re.sub(r'\n{3,}', '\n\n', content).strip()

    return title, date, content, json.dumps(attachments, ensure_ascii=False) if attachments else ''


def main():
    all_items = []

    for page in range(1, MIN_PG + 1):
        if page == 1:
            url = f'{BASE}/zwgk/tzgg/index.html'
        else:
            url = f'{BASE}/content/column/{COLUMN_ID}?pageIndex={page}'
        print(f'[页码 {page}/{MIN_PG}] {url}')
        try:
            html = http_get(url)
            items = extract_list(html)
            if not items:
                print('  → 无数据，停止')
                break
            print(f'  → 提取 {len(items)} 条')
            all_items.extend(items)
        except Exception as e:
            print(f'  → 失败: {e}')
            continue
        time.sleep(0.5)

    print(f'\n共 {len(all_items)} 条列表数据')

    # DB dedup: skip existing URLs
    import sqlite3
    db = sqlite3.connect("/root/search.db", timeout=60)
    existing_urls = set()
    try:
        cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        for row in cur.fetchall():
            existing_urls.add(row[0])
    except:
        pass
    db.close()
    before = len(all_items)
    all_items = [it for it in all_items if it["url"] not in existing_urls]
    print(f"  DB already: {len(existing_urls)} records, to fetch: {len(all_items)} items")
    if not all_items:
        print("  All already in DB - nothing to fetch")
        return



    output_path = '/tmp/hndt_output_v2.jsonl'
    count = 0
    with open(output_path, 'w', encoding='utf-8') as f:
        for i, item in enumerate(all_items, 1):
            url = item['url']
            short_title = item['title'][:40] if item['title'] else url
            print(f'[{i}/{len(all_items)}] {short_title}...', end=' ', flush=True)
            try:
                html = http_get(url)
                title, date, content, attachments = parse_detail(html)
                title = title or item['title']
                date = date or item['date']
                record = {
                    'title': title,
                    'page_url': url,
                    'content': content,
                    'publish_date': date,
                    'summary': title,
                    'site_name': SITE_NAME,
                    'tags': '通知公告',
                    'attachments': attachments
                }
                f.write(json.dumps(record, ensure_ascii=False) + '\n')
                count += 1
                print('✓')
            except Exception as e:
                print(f'✗ {e}')
                record = {
                    'title': item['title'],
                    'page_url': url,
                    'content': '',
                    'publish_date': item['date'],
                    'summary': item['title'],
                    'site_name': SITE_NAME,
                    'tags': '通知公告',
                    'attachments': ''
                }
                f.write(json.dumps(record, ensure_ascii=False) + '\n')
                count += 1
            time.sleep(0.3)

    print(f'\n✅ JSONL: {output_path} ({count} 条)')

    # Delete old garbled data first
    print('\n🗑️ 删除旧数据...')
    import sqlite3, os
    db_path = '/mnt/data/search.db'
    # Stop search service, delete old hndt records
    subprocess.run(['systemctl', 'stop', 'search_app'], capture_output=True)
    db = sqlite3.connect(db_path, timeout=60)
    deleted = db.execute("DELETE FROM gov_raw WHERE page_url LIKE '%hndt.gov.cn%'").rowcount
    db.commit()
    db.close()
    print(f'  删除了 {deleted} 条错误数据')

    # Import fresh data
    print('\n📥 导入...')
    import_script = '/root/gov_crawler/import_jsonl.py'
    r = subprocess.run(['python3', import_script, output_path],
                       capture_output=True, text=True, timeout=100, cwd='/mnt/data')
    print(r.stdout[-400:] if len(r.stdout) > 400 else r.stdout)
    if r.returncode != 0:
        print(f'⚠️ 导入异常: {r.stderr[-300:]}')

    # Rebuild FTS
    print('\n🔨 重建 FTS...')
    db = sqlite3.connect(db_path, timeout=60)
    db.execute('DELETE FROM gov_search')
    db.execute('DELETE FROM gov_search_v3')
    db.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
    # For v3: insert directly
    db.execute("INSERT INTO gov_search_v3(title, site_name, source_url, publish_date) "
               "SELECT title, site_name, page_url, publish_date FROM gov_raw")
    db.commit()
    db.close()

    # Restart
    subprocess.run(['systemctl', 'start', 'search_app'], capture_output=True, timeout=10)
    print('✅ 搜索服务已重启')
    print(f'\n✅ 完成！重新导入 {count} 条（编码已修复）')


if __name__ == '__main__':
    main()
