#!/usr/bin/env python3
"""民权县生态环境爬虫 - PowerCMS
URL: http://www.minquan.gov.cn/zwgk/jczwgk11/sthj
列表: div.mainContent > ul.infoList > li > a + span.date
分页: sthj → sthj_2 → ... → sthj_32 (10条/页)
详情: meta ArticleTitle, meta PubDate, div.conTxt
"""
import re, sys, time, os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'http://www.minquan.gov.cn/zwgk/jczwgk11/sthj'
SITE_NAME = '民权县生态环境'
GROUP = '河南'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
TIMEOUT = 30
DELAY = 1.5
DB_PATH = '/root/search.db'


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def extract_content(html):
    """Extract paragraphs, tables, and attachments from conTxt div"""
    soup = BeautifulSoup(html, 'lxml')
    con_div = soup.find('div', class_='conTxt')
    if not con_div:
        return '', []
    parts = []
    attachments = []
    for el in con_div.children:
        if isinstance(el, str):
            text = el.strip()
            if text:
                parts.append(text)
            continue
        tag = el.name.lower() if el.name else ''
        if tag == 'p':
            text = body_text(el)
            if text:
                parts.append(text)
            for a in el.find_all('a'):
                href = a.get('href', '')
                if any(ext in href.lower() for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                    attachments.append({
                        'name': a.get_text(strip=True) or os.path.basename(href),
                        'url': href
                    })
        elif tag == 'table':
            md = table_to_markdown(el)
            if md:
                parts.append(md)
        elif tag in ['ul', 'ol']:
            text = body_text(el)
            if text:
                parts.append(text)
        elif tag == 'div':
            # Recurse into div
            _walk_node(el, parts, attachments)
    parts = [p for p in parts if p.strip()]
    text_content = '\n\n'.join(parts)
    for att in attachments:
        att['url'] = urljoin(BASE_URL, att['url'])
    return text_content, attachments


def _walk_node(node, parts, attachments):
    for el in node.children:
        if isinstance(el, str):
            text = el.strip()
            if text:
                parts.append(text)
            continue
        tag = el.name.lower() if el.name else ''
        if tag == 'p':
            text = body_text(el)
            if text:
                parts.append(text)
            for a in el.find_all('a'):
                href = a.get('href', '')
                if any(ext in href.lower() for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                    attachments.append({
                        'name': a.get_text(strip=True) or os.path.basename(href),
                        'url': href
                    })
        elif tag == 'table':
            md = table_to_markdown(el)
            if md:
                parts.append(md)
        elif tag in ['ul', 'ol']:
            text = body_text(el)
            if text:
                parts.append(text)
        elif tag == 'div':
            _walk_node(el, parts, attachments)


def parse_list(html):
    """Parse list page, return list of (title, date, detail_url)"""
    soup = BeautifulSoup(html, 'lxml')
    main = soup.find('div', class_='mainContent')
    if not main:
        return []
    ul = main.find('ul', class_='infoList')
    if not ul:
        return []
    items = []
    for li in ul.find_all('li'):
        if 'noData' in li.get('class', []):
            continue
        a = li.find('a')
        if not a:
            continue
        title = a.get('title', '') or a.get_text(strip=True)
        href = a.get('href', '')
        if not href:
            continue
        full_url = urljoin(BASE_URL, href)
        span = li.find('span', class_='date')
        date_str = span.get_text(strip=True) if span else ''
        items.append((title, date_str, full_url))
    return items


def get_total_pages(html):
    """Parse last page number from pagination"""
    soup = BeautifulSoup(html, 'lxml')
    main = soup.find('div', class_='mainContent')
    if not main:
        return 1
    page = main.find('div', class_='page')
    if not page:
        return 1
    nums = []
    for a in page.find_all('a'):
        h = a.get('href', '')
        m = re.search(r'sthj_(\d+)', h)
        if m:
            nums.append(int(m.group(1)))
    return max(nums) if nums else 1


def scrape_detail(url):
    """Scrape detail page"""
    resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'lxml')
    # Title
    title = ''
    for meta in soup.find_all('meta'):
        n = meta.get('name', '') or ''
        if 'ArticleTitle' in n:
            title = (meta.get('content', '') or '').strip()
            break
    # Date
    date_str = ''
    for meta in soup.find_all('meta'):
        n = meta.get('name', '') or ''
        if 'PubDate' in n:
            date_str = (meta.get('content', '') or '').strip()[:10]
            break
    content, attachments = extract_content(resp.text)
    return title, date_str, content, attachments


def main():
    import argparse
    parser = argparse.ArgumentParser(description='民权县生态环境爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数')
    parser.add_argument('--import-db', action='store_true', default=True, help='入库')
    args = parser.parse_args()

    # Get total pages
    resp = requests.get(BASE_URL, headers=HEADERS, timeout=TIMEOUT)
    resp.encoding = 'utf-8'
    total_pages = get_total_pages(resp.text)
    pages_to_crawl = min(args.pages, total_pages)
    print(f'总页数: {total_pages}, 爬取: {pages_to_crawl}')

    all_items = parse_list(resp.text)
    print(f'列表页 1/{pages_to_crawl}: {len(all_items)} 条')

    for i in range(2, pages_to_crawl + 1):
        page_url = f'{BASE_URL}_{i}'
        print(f'列表页 {i}/{pages_to_crawl}: {page_url}')
        resp2 = requests.get(page_url, headers=HEADERS, timeout=TIMEOUT)
        resp2.encoding = 'utf-8'
        items = parse_list(resp2.text)
        print(f'  找到 {len(items)} 条')
        all_items.extend(items)
        time.sleep(DELAY)

    print(f'\n共采集 {len(all_items)} 条')
    total = len(all_items)
    inserted = 0
    skipped = 0

    for idx, (title, date_str, url) in enumerate(all_items):
        dsp = title[:40] if len(title) > 40 else title
        print(f'  [{idx+1}/{total}] {dsp}...', end=' ')
        sys.stdout.flush()
        try:
            d_title, d_date, content, attachments = scrape_detail(url)
            if not d_title:
                d_title = title
            if not content or len(content.strip()) < 10:
                print(f'跳过（正文过短）')
                skipped += 1
                continue
            print(f'✅ {len(content)}字', end='')
            if attachments:
                print(f' +{len(attachments)}附件', end='')
            print()

            if args.import_db:
                import sqlite3
                conn = sqlite3.connect(DB_PATH, timeout=60)
                c = conn.cursor()
                c.execute('SELECT id FROM gov_raw WHERE page_url=? AND site_name=?', (url, SITE_NAME))
                if c.fetchone():
                    print(f'    已存在，跳过')
                    conn.close()
                    continue
                summary = content[:200].replace('\n', ' ') if content else ''
                attach_str = '\n'.join([f"{a['name']}: {a['url']}" for a in attachments]) if attachments else ''
                c.execute('''INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, group_name, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, \'crawl_minquan_sthj.py\')''',
                          (url, d_title, content, d_date, SITE_NAME, summary, attach_str, d_date or '0000-00-00', GROUP))
                conn.commit()
                conn.close()
                inserted += 1
            time.sleep(DELAY)
        except Exception as e:
            print(f'❌ {e}')
            skipped += 1
            time.sleep(3)

    print(f'\n=== 完成 ===')
    print(f'入库: {inserted}, 跳过: {skipped}')


if __name__ == '__main__':
    main()
