#!/usr/bin/env python3
"""民权县生态环境爬虫 - PowerCMS
URL: http://www.minquan.gov.cn/zwgk/jczwgk11/sthj
列表: div.mainContent > ul.infoList > li > a + span.date
分页: sthj → sthj_2 → ... → sthj_32 (10条/页)
详情: meta ArticleTitle, meta PubDate, div.conTxt
"""
import re, sys, time, os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'http://www.minquan.gov.cn/zwgk/jczwgk11/sthj'
SITE_NAME = '民权县生态环境'
GROUP = '河南'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
TIMEOUT = 30
DELAY = 1.5
DB_PATH = '/root/search.db'


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def extract_content(html):
    """Extract paragraphs, tables, and attachments from conTxt div"""
    soup = BeautifulSoup(html, 'lxml')
    con_div = soup.find('div', class_='conTxt')
    if not con_div:
        return '', []
    parts = []
    attachments = []
    for el in con_div.children:
        if isinstance(el, str):
            text = el.strip()
            if text:
                parts.append(text)
            continue
        tag = el.name.lower() if el.name else ''
        if tag == 'p':
            text = el.get_text('\n', strip=True)
            if text:
                parts.append(text)
            for a in el.find_all('a'):
                href = a.get('href', '')
                if any(ext in href.lower() for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                    attachments.append({
                        'name': a.get_text(strip=True) or os.path.basename(href),
                        'url': href
                    })
        elif tag == 'table':
            md = table_to_markdown(el)
            if md:
                parts.append(md)
        elif tag in ['ul', 'ol']:
            text = el.get_text('\n', strip=True)
            if text:
                parts.append(text)
        elif tag == 'div':
            # Recurse into div
            _walk_node(el, parts, attachments)
    parts = [p for p in parts if p.strip()]
    text_content = '\n\n'.join(parts)
    for att in attachments:
        att['url'] = urljoin(BASE_URL, att['url'])
    return text_content, attachments


def _walk_node(node, parts, attachments):
    for el in node.children:
        if isinstance(el, str):
            text = el.strip()
            if text:
                parts.append(text)
            continue
        tag = el.name.lower() if el.name else ''
        if tag == 'p':
            text = el.get_text('\n', strip=True)
            if text:
                parts.append(text)
            for a in el.find_all('a'):
                href = a.get('href', '')
                if any(ext in href.lower() for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                    attachments.append({
                        'name': a.get_text(strip=True) or os.path.basename(href),
                        'url': href
                    })
        elif tag == 'table':
            md = table_to_markdown(el)
            if md:
                parts.append(md)
        elif tag in ['ul', 'ol']:
            text = el.get_text('\n', strip=True)
            if text:
                parts.append(text)
        elif tag == 'div':
            _walk_node(el, parts, attachments)


def parse_list(html):
    """Parse list page, return list of (title, date, detail_url)"""
    soup = BeautifulSoup(html, 'lxml')
    main = soup.find('div', class_='mainContent')
    if not main:
        return []
    ul = main.find('ul', class_='infoList')
    if not ul:
        return []
    items = []
    for li in ul.find_all('li'):
        if 'noData' in li.get('class', []):
            continue
        a = li.find('a')
        if not a:
            continue
        title = a.get('title', '') or a.get_text(strip=True)
        href = a.get('href', '')
        if not href:
            continue
        full_url = urljoin(BASE_URL, href)
        span = li.find('span', class_='date')
        date_str = span.get_text(strip=True) if span else ''
        items.append((title, date_str, full_url))
    return items


def get_total_pages(html):
    """Parse last page number from pagination"""
    soup = BeautifulSoup(html, 'lxml')
    main = soup.find('div', class_='mainContent')
    if not main:
        return 1
    page = main.find('div', class_='page')
    if not page:
        return 1
    nums = []
    for a in page.find_all('a'):
        h = a.get('href', '')
        m = re.search(r'sthj_(\d+)', h)
        if m:
            nums.append(int(m.group(1)))
    return max(nums) if nums else 1


def scrape_detail(url):
    """Scrape detail page"""
    resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'lxml')
    # Title
    title = ''
    for meta in soup.find_all('meta'):
        n = meta.get('name', '') or ''
        if 'ArticleTitle' in n:
            title = (meta.get('content', '') or '').strip()
            break
    # Date
    date_str = ''
    for meta in soup.find_all('meta'):
        n = meta.get('name', '') or ''
        if 'PubDate' in n:
            date_str = (meta.get('content', '') or '').strip()[:10]
            break
    content, attachments = extract_content(resp.text)
    return title, date_str, content, attachments


def main():
    import argparse
    parser = argparse.ArgumentParser(description='民权县生态环境爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数')
    parser.add_argument('--import-db', action='store_true', default=True, help='入库')
    args = parser.parse_args()

    # Get total pages
    resp = requests.get(BASE_URL, headers=HEADERS, timeout=TIMEOUT)
    resp.encoding = 'utf-8'
    total_pages = get_total_pages(resp.text)
    pages_to_crawl = min(args.pages, total_pages)
    print(f'总页数: {total_pages}, 爬取: {pages_to_crawl}')

    all_items = parse_list(resp.text)
    print(f'列表页 1/{pages_to_crawl}: {len(all_items)} 条')

    for i in range(2, pages_to_crawl + 1):
        page_url = f'{BASE_URL}_{i}'
        print(f'列表页 {i}/{pages_to_crawl}: {page_url}')
        resp2 = requests.get(page_url, headers=HEADERS, timeout=TIMEOUT)
        resp2.encoding = 'utf-8'
        items = parse_list(resp2.text)
        print(f'  找到 {len(items)} 条')
        all_items.extend(items)
        time.sleep(DELAY)

    print(f'\n共采集 {len(all_items)} 条')
    total = len(all_items)
    inserted = 0
    skipped = 0

    for idx, (title, date_str, url) in enumerate(all_items):
        dsp = title[:40] if len(title) > 40 else title
        print(f'  [{idx+1}/{total}] {dsp}...', end=' ')
        sys.stdout.flush()
        try:
            d_title, d_date, content, attachments = scrape_detail(url)
            if not d_title:
                d_title = title
            if not content or len(content.strip()) < 10:
                print(f'跳过（正文过短）')
                skipped += 1
                continue
            print(f'✅ {len(content)}字', end='')
            if attachments:
                print(f' +{len(attachments)}附件', end='')
            print()

            if args.import_db:
                import sqlite3
                conn = sqlite3.connect(DB_PATH)
                c = conn.cursor()
                c.execute('SELECT id FROM gov_raw WHERE page_url=? AND site_name=?', (url, SITE_NAME))
                if c.fetchone():
                    print(f'    已存在，跳过')
                    conn.close()
                    continue
                summary = content[:200].replace('\n', ' ') if content else ''
                attach_str = '\n'.join([f"{a['name']}: {a['url']}" for a in attachments]) if attachments else ''
                c.execute('''INSERT OR REPLACE INTO gov_raw
                             (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, group_name)
                             VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)''',
                          (url, d_title, content, d_date, SITE_NAME, summary, attach_str, d_date or '0000-00-00', GROUP))
                conn.commit()
                conn.close()
                inserted += 1
            time.sleep(DELAY)
        except Exception as e:
            print(f'❌ {e}')
            skipped += 1
            time.sleep(3)

    print(f'\n=== 完成 ===')
    print(f'入库: {inserted}, 跳过: {skipped}')


if __name__ == '__main__':
    main()
