#!/usr/bin/env python3
"""汨罗市建设项目环评爬虫 - ASP.NET CMS
URL: http://www.miluo.gov.cn/25308/27701/27724/27986/default.htm
列表: ul.g-list-t.list-gl > li > a + span.date
分页: default_2.htm → ... → default_75.htm (20条/页, 共1500条)
详情: meta ArticleTitle, meta PubDate, div#zoom
编码: GB2312
"""
import re, sys, time, os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'http://www.miluo.gov.cn/25308/27701/27724/27986/default.htm'
LIST_DIR = 'http://www.miluo.gov.cn/25308/27701/27724/27986/'
SITE_NAME = '汨罗市建设项目环评'
GROUP = '湖南'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
TIMEOUT = 30
DELAY = 1.5
DB_PATH = '/root/search.db'


def table_to_markdown(table):
    rows = table.find_all('tr')
    if not rows:
        return ''
    md_rows = []
    col_count = 0
    for tr in rows:
        cells = tr.find_all(['td', 'th'])
        if not cells:
            continue
        row = []
        for cell in cells:
            text = cell.get_text('\n', strip=True).replace('\n', '<br>')
            text = re.sub(r'\s+', ' ', text)
            row.append(text)
        col_count = max(col_count, len(row))
        md_rows.append(row)
    if not md_rows:
        return ''
    for row in md_rows:
        while len(row) < col_count:
            row.append('')
    lines = []
    lines.append('| ' + ' | '.join(md_rows[0]) + ' |')
    lines.append('| ' + ' | '.join(['---'] * col_count) + ' |')
    for row in md_rows[1:]:
        lines.append('| ' + ' | '.join(row) + ' |')
    return '\n'.join(lines)


def _walk_node(node, parts, attachments):
    for el in node.children:
        if isinstance(el, str):
            text = el.strip()
            if text:
                parts.append(text)
            continue
        tag = el.name.lower() if el.name else ''
        if tag == 'p':
            text = el.get_text('\n', strip=True)
            if text:
                parts.append(text)
            for a in el.find_all('a'):
                href = a.get('href', '')
                if any(ext in href.lower() for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                    attachments.append({
                        'name': a.get_text(strip=True) or os.path.basename(href),
                        'url': href
                    })
        elif tag == 'table':
            md = table_to_markdown(el)
            if md:
                parts.append(md)
        elif tag in ['ul', 'ol']:
            text = el.get_text('\n', strip=True)
            if text:
                parts.append(text)
        elif tag == 'div':
            _walk_node(el, parts, attachments)


def extract_content(html, page_url):
    """Extract content from div#zoom"""
    soup = BeautifulSoup(html, 'lxml')
    zoom = soup.find(id='zoom') or soup.find('div', class_='content_')
    if not zoom:
        return '', []
    parts = []
    attachments = []
    _walk_node(zoom, parts, attachments)
    parts = [p for p in parts if p.strip()]
    text_content = '\n\n'.join(parts)
    for att in attachments:
        att['url'] = urljoin(LIST_DIR, att['url'])
    return text_content, attachments


def parse_list(html):
    """Parse list page, return list of (title, date, detail_url)"""
    soup = BeautifulSoup(html, 'lxml')
    items = []
    for ul in soup.find_all('ul', class_='g-list-t'):
        for li in ul.find_all('li'):
            a = li.find('a')
            if not a:
                continue
            title = a.get('title', '') or a.get_text(strip=True)
            href = a.get('href', '')
            if not href:
                continue
            full_url = urljoin(LIST_DIR, href)
            span = li.find('span')
            date_str = span.get_text(strip=True) if span else ''
            # Clean date
            date_str = date_str.strip('[]【】（）()')
            items.append((title, date_str, full_url))
    return items


def get_total_pages(html):
    """Parse total page count from pagination text like '共1500条1/75页'"""
    soup = BeautifulSoup(html, 'lxml')
    page_div = soup.find('div', class_='pagenav')
    if page_div:
        text = page_div.get_text(strip=True)
        m = re.search(r'/(\d+)页', text)
        if m:
            return int(m.group(1))
    return 1


def scrape_detail(url):
    """Scrape detail page"""
    resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
    resp.encoding = 'gb2312'
    soup = BeautifulSoup(resp.text, 'lxml')
    # Title
    title = ''
    for meta in soup.find_all('meta'):
        n = meta.get('name', '') or ''
        if 'ArticleTitle' in n:
            title = (meta.get('content', '') or '').strip()
            break
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)
    # Date
    date_str = ''
    for meta in soup.find_all('meta'):
        n = meta.get('name', '') or ''
        if 'PubDate' in n:
            date_str = (meta.get('content', '') or '').strip()[:10]
            break
    content, attachments = extract_content(resp.text, url)
    return title, date_str, content, attachments


def main():
    import argparse
    parser = argparse.ArgumentParser(description='汨罗市建设项目环评爬虫')
    parser.add_argument('--pages', type=int, default=5, help='爬取页数')
    parser.add_argument('--import-db', action='store_true', default=True, help='入库')
    args = parser.parse_args()

    resp = requests.get(BASE_URL, headers=HEADERS, timeout=TIMEOUT)
    resp.encoding = 'gb2312'
    total_pages = get_total_pages(resp.text)
    pages_to_crawl = min(args.pages, total_pages)
    print(f'总页数: {total_pages}, 爬取: {pages_to_crawl}')

    all_items = parse_list(resp.text)
    print(f'列表页 1/{pages_to_crawl}: {len(all_items)} 条')

    for i in range(2, pages_to_crawl + 1):
        page_url = urljoin(LIST_DIR, f'default_{i}.htm')
        print(f'列表页 {i}/{pages_to_crawl}: {page_url}')
        resp2 = requests.get(page_url, headers=HEADERS, timeout=TIMEOUT)
        resp2.encoding = 'gb2312'
        items = parse_list(resp2.text)
        print(f'  找到 {len(items)} 条')
        all_items.extend(items)
        time.sleep(DELAY)

    print(f'\n共采集 {len(all_items)} 条')
    total = len(all_items)
    inserted = 0
    skipped = 0

    for idx, (title, date_str, url) in enumerate(all_items):
        dsp = title[:40] if len(title) > 40 else title
        print(f'  [{idx+1}/{total}] {dsp}...', end=' ')
        sys.stdout.flush()
        try:
            d_title, d_date, content, attachments = scrape_detail(url)
            if not d_title:
                d_title = title
            if not content or len(content.strip()) < 10:
                print(f'跳过（正文过短）')
                skipped += 1
                continue
            print(f'✅ {len(content)}字', end='')
            if attachments:
                print(f' +{len(attachments)}附件', end='')
            print()

            if args.import_db:
                import sqlite3
                conn = sqlite3.connect(DB_PATH)
                c = conn.cursor()
                c.execute('SELECT id FROM gov_raw WHERE page_url=? AND site_name=?', (url, SITE_NAME))
                if c.fetchone():
                    print(f'    已存在')
                    conn.close()
                    continue
                summary = content[:200].replace('\n', ' ') if content else ''
                attach_str = '\n'.join([f"{a['name']}: {a['url']}" for a in attachments]) if attachments else ''
                c.execute('''INSERT OR REPLACE INTO gov_raw
                             (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, group_name)
                             VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)''',
                          (url, d_title, content, d_date, SITE_NAME, summary, attach_str, d_date or '0000-00-00', GROUP))
                conn.commit()
                conn.close()
                inserted += 1
            time.sleep(DELAY)
        except Exception as e:
            print(f'❌ {e}')
            skipped += 1
            time.sleep(3)

    print(f'\n=== 完成 ===')
    print(f'入库: {inserted}, 跳过: {skipped}')


if __name__ == '__main__':
    main()
