#!/usr/bin/env python3
"""眉山市环评公示爬虫 - Visual Site Builder 9
URL: http://www.ms.gov.cn/zfxxgk/fdzdgknr/zdmsxx/sthj/hpgs.htm
列表: div.govnewslista245277 > table > tr[id^=line245277] > a[title] + td date
分页: hpgs.htm → hpgs/3.htm → hpgs/2.htm → hpgs/1.htm (共4页, ~50条/页)
详情: meta ArticleTitle, meta PubDate, div#vsb_content_100
"""
import re, sys, time, os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE = 'http://www.ms.gov.cn/zfxxgk/fdzdgknr/zdmsxx/sthj/'
PAGE_URLS = [
    'hpgs.htm',       # page 1
    'hpgs/3.htm',     # page 2
    'hpgs/2.htm',     # page 3
    'hpgs/1.htm',     # page 4
]
SITE_NAME = '眉山市生态环境局-环评公示'
GROUP = '四川'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
TIMEOUT = 30
DELAY = 1.5

DB_PATH = '/root/search.db'


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def _walk_node(node, parts, attachments):
    """Recurse into node children, extracting paragraphs/tables/attachments.
    Avoids double-extracting table text."""
    for el in node.children:
        if isinstance(el, str):
            text = el.strip()
            if text:
                parts.append(text)
            continue
        tag = el.name.lower() if el.name else ''
        if tag == 'p':
            text = body_text(el)
            if text:
                parts.append(text)
            for a in el.find_all('a'):
                href = a.get('href', '')
                if any(ext in href.lower() for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                    attachments.append({
                        'name': a.get_text(strip=True) or os.path.basename(href),
                        'url': href
                    })
        elif tag == 'table':
            md = table_to_markdown(el)
            if md:
                parts.append(md)
        elif tag in ['ul', 'ol']:
            text = body_text(el)
            if text:
                parts.append(text)
        elif tag == 'div':
            # Recurse into div — don't get_text() it directly (would duplicate table text)
            _walk_node(el, parts, attachments)


def extract_content(html):
    """Extract paragraphs and tables from vsb_content_100 or vsb_content div"""
    soup = BeautifulSoup(html, 'lxml')
    content_div = soup.find(id='vsb_content_100')
    if not content_div:
        content_div = soup.find(id='vsb_content')
    if not content_div:
        return '', []
    parts = []
    attachments = []
    _walk_node(content_div, parts, attachments)
    parts = [p for p in parts if p.strip()]
    text_content = '\n\n'.join(parts)
    for att in attachments:
        att['url'] = urljoin(BASE + 'hpgs.htm', att['url'])
    return text_content, attachments


def parse_list(html, base_url):
    """Parse list page, return list of (title, date, detail_url)"""
    soup = BeautifulSoup(html, 'lxml')
    div = soup.find('div', class_='govnewslista245277')
    if not div:
        print(f'  WARN: govnewslista245277 not found', file=sys.stderr)
        return []
    items = []
    for tr in div.find_all('tr', id=lambda x: x and 'line245277' in str(x)):
        a = tr.find('a')
        if not a:
            continue
        title = a.get('title', '') or a.get_text(strip=True)
        href = a.get('href', '')
        if not href:
            continue
        full_url = urljoin(base_url, href)
        tds = tr.find_all('td')
        date_str = tds[1].get_text(strip=True) if len(tds) > 1 else ''
        items.append((title, date_str, full_url))
    return items


def scrape_detail(url):
    """Scrape detail page, return (title, date, content, attachments)"""
    resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'lxml')
    title = ''
    for meta in soup.find_all('meta'):
        n = meta.get('name', '') or ''
        if 'ArticleTitle' in n:
            title = (meta.get('content', '') or '').strip()
            break
    date_str = ''
    for meta in soup.find_all('meta'):
        n = meta.get('name', '') or ''
        if 'PubDate' in n:
            date_str = (meta.get('content', '') or '').strip()[:10]
            break
    content, attachments = extract_content(resp.text)
    return title, date_str, content, attachments


def main():
    import argparse
    parser = argparse.ArgumentParser(description='眉山市环评公示爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数')
    parser.add_argument('--import-db', action='store_true', default=True, help='入库')
    args = parser.parse_args()
    pages_to_crawl = min(args.pages, len(PAGE_URLS))
    all_items = []
    for i in range(pages_to_crawl):
        page_url = urljoin(BASE, PAGE_URLS[i])
        print(f'列表页 {i+1}/{pages_to_crawl}: {page_url}')
        resp = requests.get(page_url, headers=HEADERS, timeout=TIMEOUT)
        resp.encoding = 'utf-8'
        items = parse_list(resp.text, page_url)
        print(f'  找到 {len(items)} 条')
        all_items.extend(items)
        if i < pages_to_crawl - 1:
            time.sleep(DELAY)
    print(f'\n共采集 {len(all_items)} 条列表数据')
    total = len(all_items)
    inserted = 0
    skipped = 0
    for idx, (title, date_str, url) in enumerate(all_items):
        dsp = title[:40] if len(title) > 40 else title
        print(f'  [{idx+1}/{total}] {dsp}...', end=' ')
        sys.stdout.flush()
        try:
            d_title, d_date, content, attachments = scrape_detail(url)
            if not d_title:
                d_title = title
            if not content or len(content.strip()) < 10:
                print(f'跳过（正文过短）')
                skipped += 1
                continue
            print(f'✅ {len(content)}字', end='')
            if attachments:
                print(f' +{len(attachments)}附件', end='')
            print()
            if args.import_db:
                import sqlite3
                conn = sqlite3.connect(DB_PATH, timeout=60)
                c = conn.cursor()
                c.execute('SELECT id FROM gov_raw WHERE page_url=? AND site_name=?', (url, SITE_NAME))
                if c.fetchone():
                    print(f'    已存在，跳过')
                    conn.close()
                    continue
                summary = content[:200].replace('\n', ' ') if content else ''
                attach_str = '\n'.join([f"{a['name']}: {a['url']}" for a in attachments]) if attachments else ''
                c.execute('''INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, group_name, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, \'crawl_ms_hpgs.py\')''',
                          (url, d_title, content, d_date, SITE_NAME, summary, attach_str, d_date or '0000-00-00', GROUP))
                conn.commit()
                conn.close()
                inserted += 1
            time.sleep(DELAY)
        except Exception as e:
            print(f'❌ {e}')
            skipped += 1
            time.sleep(3)
    print(f'\n=== 完成 ===')
    print(f'入库: {inserted}, 跳过: {skipped}')


if __name__ == '__main__':
    main()
