#!/usr/bin/env python3
"""
Crawl 瓜州县人民政府 - 公示公告
https://www.guazhou.gov.cn/guazhou/c109632/tab2.shtml
API: /common/search/{channelId}?_isJson=true&_pageSize=15&page=N
Detail: <meta name="ArticleTitle"> + <div class="Articlecontent-div">content</div>
Date: 日期：YYYY-MM-DD HH:MM
"""

import requests, json, re, os, sys, time
from bs4 import BeautifulSoup
# 支持 --pages 参数
import argparse as _AP
_AP_PARSER = _AP.ArgumentParser()
_AP_PARSER.add_argument("--pages", type=int, default=0, help="限制页数")
_AP_ARGS, _ = _AP_PARSER.parse_known_args()
_MAX_PAGES_ARG = _AP_ARGS.pages

CHANNEL_ID = '8b14be659dcc464e8feac3deb07f4706'
API_BASE = 'https://www.guazhou.gov.cn/common/search/'

OUTPUT_FILE = '/root/gov_crawler/output/guazhou.jsonl'
os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)

session = requests.Session()
session.headers.update({
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Accept': 'application/json, text/plain, */*',
})

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def fetch_list(page=1, page_size=15):
    url = f'{API_BASE}{CHANNEL_ID}?_isAgg=false&_isJson=true&_pageSize={page_size}&_template=index&_rangeTimeGte=&_channelName=&page={page}'
    resp = session.get(url, timeout=30)
    return resp.json()

def parse_detail(url, html):
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from meta ArticleTitle
    title = ''
    meta = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta and meta.get('content'):
        title = meta['content'].strip()
    
    # Date from 日期： text
    pub_date = ''
    m = re.search(r'日期[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
    if m:
        pub_date = m.group(1)
    
    # Content from Articlecontent-div
    content_div = soup.find('div', class_='Articlecontent-div')
    content_html = ''
    summary = ''
    attachments = []
    if content_div:
        # Keep tables but strip inline styles
        for tag in content_div.find_all(True):
            keep = ['href', 'src', 'alt', 'target']
            for attr in list(tag.attrs):
                if attr not in keep:
                    del tag[attr]
        content_html = str(content_div)
        summary = body_text(content_div)
        
        # Attachments
        for a in content_div.find_all('a', href=True):
            href = a['href']
            if any(href.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                full_url = href if href.startswith('http') else f'https://www.guazhou.gov.cn{href}'
                attachments.append({
                    'url': full_url,
                    'text': a.get_text(strip=True) or os.path.basename(href)
                })
    
    return {
        'title': title,
        'pub_date': pub_date,
        'content_html': content_html,
        'summary': summary[:500] if summary else '',
        'attachments': attachments
    }

def crawl_all():
    # Get first page to determine total
    data = fetch_list(1)
    total = data.get('data', {}).get('total', 0)
    total_pages = (total + 14) // 15
    print(f'Total items: {total}, Total pages: {total_pages}')
    
    # Gather all list items from all pages
    all_items = []
    limit = _MAX_PAGES_ARG if _MAX_PAGES_ARG > 0 else total_pages
    if limit < total_pages:
        print(f'Limiting to first {limit} page(s) of {total_pages} (--pages {_MAX_PAGES_ARG})')
    for page in range(1, min(limit, total_pages) + 1):
        try:
            data = fetch_list(page)
            results = data.get('data', {}).get('results', [])
            for r in results:
                all_items.append({
                    'url': r.get('url', ''),
                    'title': r.get('title', ''),
                    'date': (r.get('publishedTimeStr', '') or '')[:10],
                })
            print(f'  Page {page}/{total_pages}: {len(results)} items')
        except Exception as e:
            print(f'  [ERROR] Page {page}: {e}')
        time.sleep(0.1)
    
    print(f'\nTotal list items: {len(all_items)}')
    
    # Crawl details
    record_count = 0
    with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:
        for i, item in enumerate(all_items):
            print(f'  [{i+1}/{len(all_items)}] {item["title"][:40]}...')
            try:
                resp = session.get(item['url'], timeout=30)
                resp.encoding = 'utf-8'
                detail = parse_detail(item['url'], resp.text)
                
                title = detail['title'] or item['title']
                
                record = {
                    'title': title,
                    'url': item['url'],
                    'date': detail['pub_date'] or item['date'],
                    'content': detail['content_html'],
                    'summary': detail['summary'],
                    'site_name': '瓜州县人民政府',
                    'group': '公示公告',
                    'attachments': json.dumps(detail['attachments'], ensure_ascii=False) if detail['attachments'] else '',
                }
                f.write(json.dumps(record, ensure_ascii=False) + '\n')
                record_count += 1
            except Exception as e:
                print(f'    [ERROR] {item["url"]}: {e}')
            time.sleep(0.1)
    
    print(f'\nDone! {record_count} records written to {OUTPUT_FILE}')

if __name__ == '__main__':
    is_inc = '--incremental' in sys.argv
    if is_inc:
        print('Incremental mode: page 1 only')
        data = fetch_list(1)
        results = data.get('data', {}).get('results', [])
        print(f'Found {len(results)} items')
        with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:
            for r in results:
                try:
                    resp = session.get(r['url'], timeout=30)
                    resp.encoding = 'utf-8'
                    detail = parse_detail(r['url'], resp.text)
                    title = detail['title'] or r.get('title', '')
                    record = {
                        'title': title,
                        'url': r['url'],
                        'date': detail['pub_date'] or (r.get('publishedTimeStr', '') or '')[:10],
                        'content': detail['content_html'],
                        'summary': detail['summary'],
                        'site_name': '瓜州县人民政府',
                        'group': '公示公告',
                        'attachments': json.dumps(detail['attachments'], ensure_ascii=False) if detail['attachments'] else '',
                    }
                    f.write(json.dumps(record, ensure_ascii=False) + '\n')
                except Exception as e:
                    print(f'ERROR: {e}')
                time.sleep(0.1)
        print(f'Incremental done: {len(results)} records')
    else:
        crawl_all()
