#!/usr/bin/env python3
"""延长县人民政府 - 公示公告 爬虫
URL: https://www.yanchangxian.gov.cn/xwzx/gsgg/{page}.html
系统: 延长自有CMS (自定义)
详情容器: div.m-txt-article → <p>段落
"""

import requests
import json
import sys
import os
import re
import time
import sqlite3
from bs4 import BeautifulSoup
from datetime import datetime

BASE_URL = 'https://www.yanchangxian.gov.cn'
DOMAIN = 'www.yanchangxian.gov.cn'
SITE = '延长县人民政府-公示公告'
COLUMN = '公示公告'
PROVINCE = '陕西'
PER_PAGE = 20
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Referer': BASE_URL + '/xwzx/gsgg/1.html'
}
session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def fetch_list(page):
    """Fetch list page, return list of (title, url, date)"""
    url = f'{BASE_URL}/xwzx/gsgg/{page}.html'
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'html.parser')

        items = []
        ul = soup.select_one('div.m-lst36 ul')
        if not ul:
            log(f'  [WARN] No m-lst36 ul found on page {page}')
            return items

        for li in ul.find_all('li', recursive=False):
            a = li.find('a')
            span = li.find('span')
            if a:
                href = a.get('href', '')
                title = a.get('title') or a.get_text(strip=True)
                date = span.get_text(strip=True) if span else ''
                if href.startswith('/'):
                    href = BASE_URL + href
                items.append((title, href, date))

        return items
    except Exception as e:
        log(f'  [ERR] List page {page} failed: {e}')
        return []


def fetch_detail(url):
    """Fetch detail page, return (title, date, content_text, attachments)"""
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        log(f'  [ERR] Detail fetch failed: {url} - {e}')
        return None

    soup = BeautifulSoup(r.text, 'html.parser')

    # Title - from h1 or meta
    title = ''
    h1 = soup.find('h1')
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        mt = soup.find('meta', attrs={'name': re.compile(r'ArticleTitle|ContentTitle', re.I)})
        if mt and mt.get('content'):
            title = mt['content'].strip()
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            title = title_tag.get_text(strip=True).replace('--延长县人民政府', '').strip()

    # Date - from meta pubdate
    pubdate = ''
    meta_pub = soup.find('meta', attrs={'name': re.compile(r'pubdate|publicdate|publishdate', re.I)})
    if meta_pub and meta_pub.get('content'):
        pubdate = meta_pub['content'].strip()[:10]
    if not pubdate:
        date_match = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', soup.get_text())
        if date_match:
            pubdate = date_match.group(1).replace('/', '-')

    # Content - div.m-txt-article → <p> paragraphs
    content_div = soup.select_one('div.m-txt-article')
    if not content_div:
        content_div = soup.select_one('div.m-lst36-article')

    content_parts = []
    attachments = []

    if content_div:
        # Extract text from direct <p> children
        for child in content_div.find_all('p', recursive=False):
            # Check for attachments inside this p
            for a in child.find_all('a'):
                href = a.get('href', '')
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href.lower()):
                    attach_text = a.get_text(strip=True) or os.path.basename(href)
                    full_url = href if href.startswith('http') else (BASE_URL + href if href.startswith('/') else url + '/' + href)
                    if not any(att['url'] == full_url for att in attachments):
                        attachments.append({'url': full_url, 'name': attach_text})

            # Get clean text
            text = body_text(child)
            if text:
                content_parts.append(text)

        # Also handle tables
        for table in content_div.find_all('table'):
            table_html = str(table)
            content_parts.append(f'[表格]\n{table_html}\n[/表格]')

    content_text = '\n\n'.join(content_parts)

    # Also find attachments across the whole page
    for a in soup.find_all('a', href=True):
        href = a['href']
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps)$', href.lower()):
            attach_text = a.get_text(strip=True) or os.path.basename(href)
            full_url = href if href.startswith('http') else (BASE_URL + href if href.startswith('/') else url + '/' + href)
            if not any(att['url'] == full_url for att in attachments):
                attachments.append({'url': full_url, 'name': attach_text})

    return title, pubdate, content_text, attachments


def import_to_db(record):
    try:
        db = sqlite3.connect(DB_PATH, timeout=10)
        title = (record.get("title") or "")[:500]
        page_url = (record.get("page_url") or "")[:1000]
        content = record.get("content") or ""
        publish_date = (record.get("publish_date") or "")[:20]
        site_name = (record.get("site_name") or "unknown")[:100]
        attachments_str = json.dumps(record.get("attachments") or [], ensure_ascii=False)

        old = db.execute("SELECT rowid FROM gov_raw WHERE page_url = ?", (page_url,)).fetchone()
        if old:
            db.execute("DELETE FROM gov_search WHERE rowid = ?", (old[0],))
            db.execute("DELETE FROM gov_raw WHERE page_url = ?", (page_url,))

        db.execute(
            "INSERT INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?, ?, ?, ?, ?, ?, 'synced', ?)",
            (title, page_url, content, publish_date, site_name, page_url, attachments_str)
        )
        new_rowid = db.execute("SELECT last_insert_rowid()").fetchone()[0]
        summary = content[:500] if content else title[:500]
        # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
        #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
        db.commit()
        db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                   (new_rowid, title, site_name, summary))
        db.commit()
        db.close()
        log(f"  ✅ {title[:30]}")
        return True
    except Exception as e:
        log(f"  [ERR] DB import failed for {record.get('title','')}: {e}")
        return False


def main():
    import argparse
    parser = argparse.ArgumentParser(description='延长县人民政府-公示公告爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取前N页（每页20条，默认5页=100条）')
    args = parser.parse_args()

    count = 0
    for page in range(1, args.pages + 1):
        log(f'📄 第{page}页...')
        items = fetch_list(page)
        if not items:
            log(f'  → 无更多数据，结束')
            break

        log(f'  → {len(items)} 条')

        for idx, (title, url, date) in enumerate(items, 1):
            log(f'  ({idx}/{len(items)}) {title[:40]}...')

            try:
                result = fetch_detail(url)
                if not result:
                    continue

                detail_title, pubdate, content, attachments = result
                final_title = detail_title or title
                final_date = pubdate or date

                record = {
                    'title': final_title,
                    'page_url': url,
                    'publish_date': final_date,
                    'content': content,
                    'attachments': attachments,
                    'site_name': SITE,
                    'column': COLUMN,
                    'province': PROVINCE,
                }

                ok = import_to_db(record)
                if ok:
                    count += 1
            except Exception as e:
                log(f'  [ERR] 详情页处理失败: {url} - {e}')

            time.sleep(0.3)

        if page < args.pages:
            time.sleep(1.5)

    log(f'\n✅ {SITE} 爬取完成，共入库 {count} 条')


if __name__ == '__main__':
    main()
