#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
凤阳县人民政府 - 环评公示爬虫
CMS: Lonsun（蓝汛）
列表: ul.doc_list > li > a[href] + span
分页: /content/column/161694530?pageIndex=N（48页，20条/页）
详情: div.j-fontContent（文本+<br>结构，无<p>/<table>）
"""

import re
import sys
import json
import time
import requests
import sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://www.fengyang.gov.cn'
LIST_URL_P1 = BASE_URL + '/zwdt/ztzl/rdzt/hpgs/index.html'
LIST_URL_PN = BASE_URL + '/content/column/161694530?pageIndex={}'
DB_PATH = '/root/search.db'
SITE_NAME = '凤阳县生态环境分局-环评公示'
CATEGORY = '安徽'
GROUP = '安徽'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

MAX_PAGES = 5


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute('PRAGMA journal_mode=WAL')
    conn.execute('PRAGMA busy_timeout=5000')
    return conn


def get_soup(url, session=None):
    s = session or requests.Session()
    try:
        resp = s.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return None
        return BeautifulSoup(resp.text, 'html.parser')
    except Exception:
        return None


def extract_list_items(soup):
    items = []
    ul = soup.find('ul', class_=lambda x: x and 'doc_list' in str(x))
    if not ul:
        return items
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        title = (a.get('title') or a.get_text(strip=True) or '').strip()
        if not title:
            continue
        href = a['href'].strip()
        full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
        span = li.find('span')
        date = span.get_text(strip=True) if span else ''
        items.append({'title': title, 'url': full_url, 'date': date})
    return items


def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return None, None, None, None, None, None
    except Exception as e:
        print('  [ERROR] fetch detail: %s' % e, file=sys.stderr)
        return None, None, None, None, None, None

    soup = BeautifulSoup(r.text, 'html.parser')

    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()

    publish_date = ''
    meta_pub = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_pub and meta_pub.get('content'):
        publish_date = meta_pub['content'].strip()[:10]

    source_url = ''
    meta_src = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_src and meta_src.get('content'):
        source_url = meta_src['content'].strip()

    content_div = soup.find('div', class_='j-fontContent') or soup.find('div', class_='newscontnet')
    content_parts = []
    attachments = []
    images = []

    if content_div:
        for noise in content_div.find_all(['script', 'style']):
            noise.decompose()

        # 该CMS正文使用文本节点+<br>+<img>+<a>直接排列
        for child in content_div.children:
            if child.name is None:
                text = child.strip()
                if text and len(text) > 2:
                    content_parts.append(text)
            elif child.name == 'br':
                continue  # <br>只是换行，靠\n\n分组
            elif child.name == 'img':
                src = child.get('src', '')
                alt = child.get('alt', '')
                if src:
                    if not src.startswith('http'):
                        src = urljoin(url, src)
                    images.append(src)
                    content_parts.append('![%s](%s)' % (alt or '', src))
            elif child.name == 'a':
                txt = child.get_text(strip=True)
                href = child.get('href', '')
                if txt and href:
                    if not href.startswith('http'):
                        href = urljoin(url, href)
                    # 附件链接
                    if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                        attachments.append({'name': txt, 'url': href})
                    content_parts.append('[%s](%s)' % (txt, href))
            elif child.name in ('p', 'div'):
                txt = child.get_text(' ', strip=True)
                if txt and len(txt) > 2:
                    # 检查是否包含附件链接
                    for a in child.find_all('a', href=True):
                        ah = a['href'].strip()
                        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', ah, re.I):
                            full_ah = ah if ah.startswith('http') else urljoin(url, ah)
                            attachments.append({
                                'name': a.get_text(strip=True) or ah.split('/')[-1],
                                'url': full_ah,
                            })
                    # 检查是否包含img
                    for img in child.find_all('img'):
                        src = img.get('src', '')
                        alt = img.get('alt', '')
                        if src:
                            if not src.startswith('http'):
                                src = urljoin(url, src)
                            content_parts.append('![%s](%s)' % (alt or '', src))
                    content_parts.append(txt)

    content = '\n\n'.join(content_parts)

    # 降级处理
    if len(content.strip()) < 20:
        content = '<p><a href="%s">%s</a></p>' % (url, title or url.split('/')[-1])
        for img_src in images:
            content += '\n![](%s)' % img_src

    summary = content[:200] if len(content) > 200 else content
    attach_json = json.dumps(attachments, ensure_ascii=False) if attachments else ''

    return title, publish_date, source_url, content, summary, attach_json


def main():
    import argparse
    parser = argparse.ArgumentParser(description='凤阳县环评公示爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES)
    parser.add_argument('--full', action='store_true')
    args = parser.parse_args()

    max_pages = 48 if args.full else args.max_pages

    session = requests.Session()
    all_items = []
    seen_urls = set()

    for page in range(1, max_pages + 1):
        url = LIST_URL_P1 if page == 1 else LIST_URL_PN.format(page)
        print('列表页 %d/%d: %s' % (page, max_pages, url))
        soup = get_soup(url, session)
        if not soup:
            print('  FAILED')
            continue

        items = extract_list_items(soup)
        if not items:
            print('  空列表，停止')
            break

        new_count = 0
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
                new_count += 1

        print('  本页%d条，新增%d条，累计%d条' % (len(items), new_count, len(all_items)))
        if new_count == 0:
            break
        time.sleep(0.5)

    print('\n共%d篇文章，开始抓取详情...' % len(all_items))
    conn = init_db()
    inserted = 0
    errors = 0

    INSERT_SQL = '''INSERT OR IGNORE INTO gov_raw
        (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
        VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)'''

    for idx, item in enumerate(all_items):
        try:
            print('  [%d/%d] %s...' % (idx + 1, len(all_items), item['title'][:40]))
            title, pub_date, src_url, content, summary, attachments = extract_detail(item['url'])
            if title is None:
                errors += 1
                continue

            final_title = title or item['title']
            final_date = pub_date or item['date']
            final_summary = summary or content[:200] if content else final_title

            conn.execute(INSERT_SQL, (
                SITE_NAME, src_url or item['url'], item['url'],
                final_title.strip(), final_date,
                final_summary.strip(), content,
                CATEGORY, attachments, GROUP,
            ))
            conn.commit()
            inserted += 1
            time.sleep(0.3)

        except Exception as e:
            print('  ERROR: %s' % e)
            errors += 1

    conn.close()
    print('\n完成！新增%d条，错误%d条' % (inserted, errors))
    return inserted


if __name__ == '__main__':
    main()
