#!/usr/bin/env python3
"""阳江市人民政府 - 公告公示 爬虫
URL: http://www.yangjiang.gov.cn/yj/ywdt/gggs/index.html → index_{page}.html
系统: 广东省统一政府CMS (NFCMS)
详情容器: div.zw#zoomcon → <p>段落，div.fj → 附件
"""

import requests
import json
import sys
import os
import re
import time
import sqlite3
from bs4 import BeautifulSoup

BASE_URL = 'http://www.yangjiang.gov.cn'
DOMAIN = 'www.yangjiang.gov.cn'
SITE = '阳江市人民政府-公告公示'
COLUMN = '公告公示'
PROVINCE = '广东'
PER_PAGE = 20
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Referer': BASE_URL + '/yj/ywdt/gggs/'
}
session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def log(msg):
    print(msg, file=sys.stderr, flush=True)


def fetch_list(page):
    """Fetch list page, return list of (title, url, date)"""
    if page == 1:
        url = f'{BASE_URL}/yj/ywdt/gggs/index.html'
    else:
        url = f'{BASE_URL}/yj/ywdt/gggs/index_{page}.html'
    
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'html.parser')
        
        items = []
        ul = soup.select_one('ul.list')
        if not ul:
            log(f'  [WARN] No ul.list on page {page}')
            return items
        
        for li in ul.find_all('li'):
            a = li.find('a')
            span = li.find('span', class_='time')
            if a:
                href = a.get('href', '')
                title = a.get('title') or a.get_text(strip=True)
                date = span.get_text(strip=True) if span else ''
                # Only include internal content pages, skip hdjlpt links
                if '/yj/ywdt/gggs/content/' not in href:
                    log(f'  [SKIP] External/different domain link: {href[:60]}')
                    continue
                if not href.startswith('http'):
                    href = BASE_URL + href
                items.append((title, href, date))
        
        return items
    except Exception as e:
        log(f'  [ERR] List page {page} failed: {e}')
        return []


def fetch_detail(url):
    """Fetch detail page, return (title, date, content_text, attachments)"""
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
    except Exception as e:
        log(f'  [ERR] Detail fetch failed: {url} - {e}')
        return None
    
    soup = BeautifulSoup(r.text, 'html.parser')
    article = soup.select_one('div.acticle')
    if not article:
        log(f'  [WARN] No div.acticle found: {url}')
        return None
    
    # Title - from h3 inside article
    title = ''
    h3 = article.find('h3')
    if h3:
        title = h3.get_text(strip=True)
    if not title:
        mt = soup.find('meta', attrs={'name': re.compile(r'ArticleTitle', re.I)})
        if mt and mt.get('content'):
            title = mt['content'].strip()
    
    # Date - from span.time inside zw-info
    pubdate = ''
    time_span = article.select_one('span.time')
    if time_span:
        time_text = time_span.get_text(strip=True)
        date_match = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', time_text)
        if date_match:
            pubdate = date_match.group(1).replace('/', '-')
    if not pubdate:
        meta_pub = soup.find('meta', attrs={'name': re.compile(r'pubdate|PubDate', re.I)})
        if meta_pub and meta_pub.get('content'):
            pubdate = meta_pub['content'].strip()[:10]
    
    # Content - div.zw#zoomcon → <p>
    content_parts = []
    attachments = []
    
    zoom = article.select_one('div.zw#zoomcon')
    if zoom:
        for child in zoom.find_all('p', recursive=False):
            text = body_text(child)
            if text:
                content_parts.append(text)
        
        # Also check for tables
        for table in zoom.find_all('table'):
            content_parts.append(f'[表格]\n{str(table)}\n[/表格]')
    
    # Attachments - div.fj
    fj = article.select_one('div.fj')
    if fj:
        for a in fj.find_all('a', class_='file'):
            href = a.get('href', '')
            name = a.get_text(strip=True)
            if href:
                if not href.startswith('http'):
                    href = BASE_URL + href
                if not any(att['url'] == href for att in attachments):
                    attachments.append({'url': href, 'name': name or '附件'})
    
    # Also find other attachment links in zoom
    if zoom:
        for a in zoom.find_all('a', href=True):
            h = a['href']
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|et)$', h.lower()):
                full_url = h if h.startswith('http') else (BASE_URL + h if h.startswith('/') else '')
                if full_url and not any(att['url'] == full_url for att in attachments):
                    attachments.append({'url': full_url, 'name': a.get_text(strip=True) or '附件'})
    
    content_text = '\n\n'.join(content_parts)
    
    return title, pubdate, content_text, attachments


def import_to_db(record):
    try:
        db = sqlite3.connect(DB_PATH, timeout=10)
        title = (record.get("title") or "")[:500]
        page_url = (record.get("page_url") or "")[:1000]
        content = record.get("content") or ""
        publish_date = (record.get("publish_date") or "")[:20]
        site_name = (record.get("site_name") or "unknown")[:100]
        attachments_str = json.dumps(record.get("attachments") or [], ensure_ascii=False)

        old = db.execute("SELECT rowid FROM gov_raw WHERE page_url = ?", (page_url,)).fetchone()
        if old:
            db.execute("DELETE FROM gov_search WHERE rowid = ?", (old[0],))
            db.execute("DELETE FROM gov_raw WHERE page_url = ?", (page_url,))

        db.execute(
            "INSERT INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?, ?, ?, ?, ?, ?, 'synced', ?)",
            (title, page_url, content, publish_date, site_name, page_url, attachments_str)
        )
        new_rowid = db.execute("SELECT last_insert_rowid()").fetchone()[0]
        summary = content[:500] if content else title[:500]
        db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                   (new_rowid, title, site_name, summary))
        db.commit()
        db.close()
        log(f"  ✅ {title[:30]}")
        return True
    except Exception as e:
        log(f"  [ERR] DB import failed for {record.get('title','')}: {e}")
        return False


def main():
    import argparse
    parser = argparse.ArgumentParser(description='阳江市人民政府-公告公示爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取前N页（每页约20条，默认5页=~100条）')
    args = parser.parse_args()

    count = 0
    for page in range(1, args.pages + 1):
        log(f'📄 第{page}页...')
        items = fetch_list(page)
        if not items:
            log(f'  → 无更多数据，结束')
            break
        
        log(f'  → {len(items)} 条')

        for idx, (title, url, date) in enumerate(items, 1):
            log(f'  ({idx}/{len(items)}) {title[:40]}...')

            try:
                result = fetch_detail(url)
                if not result:
                    log(f'  [SKIP] Could not fetch detail')
                    continue

                detail_title, pubdate, content, attachments = result
                final_title = detail_title or title
                final_date = pubdate or date

                record = {
                    'title': final_title,
                    'page_url': url,
                    'publish_date': final_date,
                    'content': content,
                    'attachments': attachments,
                    'site_name': SITE,
                    'column': COLUMN,
                    'province': PROVINCE,
                }

                ok = import_to_db(record)
                if ok:
                    count += 1
            except Exception as e:
                log(f'  [ERR] 详情页处理失败: {url} - {e}')

            time.sleep(0.3)

        if page < args.pages:
            time.sleep(1.5)

    log(f'\n✅ {SITE} 爬取完成，共入库 {count} 条')


if __name__ == '__main__':
    main()
