#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
鄂尔多斯市人民政府 - 公示公告爬虫
URL: http://www.ordos.gov.cn/xw_127672/gsgg/
CMS: TRS
List: ul.yzgl_right_list > li > a[href] + span
Detail: div.xl_bt (title), div.xl_time (date), div.TRS_Editor (content)
Pagination: index.html (p1), index_1.html (p2)...index_6.html (p7) 13/page ~91
"""

import requests
import re
import sqlite3
import sys
import os
from urllib.parse import urljoin
from bs4 import BeautifulSoup
from datetime import datetime

DB_PATH = '/root/search.db'
BASE_URL = 'http://www.ordos.gov.cn/xw_127672/gsgg'
SITE_NAME = '鄂尔多斯市-公示公告'
GROUP = '内蒙古'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}
TOTAL_PAGES = 7  # countPage = 7

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_soup(url, timeout=15):
    r = requests.get(url, headers=HEADERS, timeout=timeout)
    r.encoding = 'utf-8'
    return BeautifulSoup(r.text, 'html.parser')

def extract_title(soup):
    div = soup.select_one('div.xl_bt')
    if div:
        return div.get_text(strip=True)
    return ''

def extract_date(soup):
    time_div = soup.select_one('div.xl_time')
    if time_div:
        m = re.search(r'发布时间[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})', time_div.get_text())
        if m:
            return m.group(1).replace('/', '-')
    return ''

def extract_content(soup):
    """从 div.TRS_Editor 提取正文"""
    editor = soup.select_one('div#article div.TRS_Editor')
    if not editor:
        editor = soup.select_one('div.xl_zw div.TRS_Editor')
    if not editor:
        return '', []

    for s in editor.find_all(['script', 'style']):
        s.decompose()

    parts = []
    attachments = []
    
    for el in list(editor.children):
        if el.name is None:
            txt = el.strip()
            if txt:
                parts.append(txt)
            continue
        tag = el.name.lower()
        if tag == 'p':
            txt = body_text(el)
            if not txt:
                continue
            parts.append(txt)
        elif tag == 'table':
            if el.find_parent('table'):
                continue
            parts.append(str(el))
        elif tag in ('div', 'section'):
            # Check for table inside wrapper div
            tbl = el.find('table', recursive=False)
            if tbl:
                parts.append(str(tbl))
            else:
                txt = body_text(el)
                if txt:
                    parts.append(txt)

    # Global attachment scan across entire editor
    for a in editor.find_all('a', href=True):
        href = a['href']
        if re.search(r'\.(doc|docx|xls|xlsx|pdf|zip|rar|txt)$', href, re.I):
            full_url = urljoin(f"{BASE_URL}/", href)
            title = a.get_text(strip=True) or os.path.basename(full_url)
            # Dedup
            if not any(x['url'] == full_url for x in attachments):
                attachments.append({'url': full_url, 'title': title})
    return '\n\n'.join(parts), attachments

def crawl_page(page_num):
    """爬取单页列表"""
    if page_num == 1:
        url = f"{BASE_URL}/index.html"
    else:
        url = f"{BASE_URL}/index_{page_num-1}.html"

    print(f"  列表页: {url}")
    soup = get_soup(url)
    items = []

    ul = soup.select_one('ul.yzgl_right_list')
    if not ul:
        print(f"  × 未找到列表 ul.yzgl_right_list")
        return items

    for li in ul.find_all('li', recursive=False):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href']
        title = a.get_text(strip=True)
        if not title:
            continue

        span = li.find('span')
        date_str = span.get_text(strip=True) if span else ''

        full_url = urljoin(f"{BASE_URL}/", href)
        items.append({
            'url': full_url,
            'title': title,
            'date': date_str,
        })
    return items

def crawl(max_pages=7):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    page_count = min(max_pages, TOTAL_PAGES)
    total = 0
    empty = 0
    has_body = 0
    total_segments = 0
    total_attachments = 0

    for p in range(1, page_count + 1):
        items = crawl_page(p)
        for item in items:
            total += 1
            print(f"  [{total}] {item['title'][:40]}...")
            try:
                detail_soup = get_soup(item['url'])
                title = extract_title(detail_soup) or item['title']
                date = extract_date(detail_soup) or item['date']
                content, attachments = extract_content(detail_soup)

                segments = len(content.split('\n\n')) if content else 0
                has_body_flag = 1 if content.strip() else 0

                attachments_json = str([a['title'] for a in attachments]) if attachments else ''
                attachments_str = '\n'.join([f"[{a['title']}]({a['url']})" for a in attachments])

                content_with_attachments = content
                if attachments_str:
                    content_with_attachments += '\n\n' + attachments_str

                c.execute('''INSERT OR REPLACE INTO gov_raw
                    (page_url, site_name, title, date_rank, content, summary, publish_date, attachments)
                    VALUES (?,?,?,?,?,?,?,?)''', (
                    item['url'],
                    SITE_NAME,
                    title,
                    date.replace('-', '') if date else '',
                    content_with_attachments,
                    content[:200] if content else '',
                    date,
                    attachments_json,
                ))
                conn.commit()

                if has_body_flag:
                    has_body += 1
                else:
                    empty += 1
                total_segments += segments
                total_attachments += len(attachments)

            except Exception as e:
                print(f"  × 详情页错误: {item['url']} - {e}")
                empty += 1

    conn.close()
    return total, empty, has_body, total_segments, total_attachments

if __name__ == '__main__':
    max_pages = 7
    if '--pages' in sys.argv:
        idx = sys.argv.index('--pages')
        if idx + 1 < len(sys.argv):
            max_pages = int(sys.argv[idx + 1])

    total, empty, has_body, total_segments, total_attachments = crawl(max_pages)
    print(f"\n===== 鄂尔多斯市 爬取完成 =====")
    print(f"共爬取: {total} 条")
    print(f"有正文: {has_body} 条 ({has_body/total*100:.1f}%)")
    print(f"空正文: {empty} 条 ({empty/total*100:.1f}%)")
    print(f"平均段落数: {total_segments/max(has_body,1):.1f}")
    print(f"附件数: {total_attachments}")
