#!/usr/bin/env python3
"""
当涂县人民政府 - 公示公告爬虫
CMS: Lonsun（蓝汛）
列表: ul.doc_list > li > a[title] + span.right.date
分页: /content/column/11296787?pageIndex=N （44页，20条/页，实际24条/页）
详情: div.newscontnet.minh500 + meta ArticleTitle/PubDate
"""

import re
import sys
import json
import time
import hashlib
import sqlite3
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://www.dangtu.gov.cn'
LIST_URL_P1 = BASE_URL + '/xwzx/gsgg/index.html'
LIST_URL_PN = BASE_URL + '/content/column/11296787?pageIndex={}'
DB_PATH = '/root/search.db'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
    'Referer': BASE_URL + '/xwzx/gsgg/',
}

MAX_PAGES = 5  # 默认前5页
MAX_PAGES_FULL = 44  # 全量44页


def get_soup(url, session=None):
    s = session or requests.Session()
    resp = s.get(url, headers=HEADERS, timeout=30)
    resp.encoding = 'utf-8'
    if resp.status_code != 200:
        print(f'  WARNING: HTTP {resp.status_code} for {url}')
        return None
    return BeautifulSoup(resp.text, 'html.parser')


def extract_list_page(soup):
    """从列表页提取文章链接和日期"""
    items = []
    ul = soup.find('ul', class_='doc_list')
    if not ul:
        return items
    for li in ul.find_all('li'):
        a = li.find('a')
        span = li.find('span', class_='right')
        if a and a.get('href'):
            title = a.get('title', '').strip() or a.get_text(strip=True)
            href = a['href'].strip()
            if not href.startswith('http'):
                href = urljoin(BASE_URL, href)
            date = ''
            if span:
                date = span.get_text(strip=True)
            if title:
                items.append({'title': title, 'url': href, 'date': date})
    return items


def extract_detail(soup, url):
    """提取详情页内容"""
    # 标题
    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()

    # 发布日期
    publish_date = ''
    meta_pub = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_pub and meta_pub.get('content'):
        publish_date = meta_pub['content'].strip()[:10]

    # 来源
    content_source = ''
    meta_src = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_src and meta_src.get('content'):
        content_source = meta_src['content'].strip()

    # 正文
    content_div = soup.find('div', class_='newscontnet')
    if not content_div:
        content_div = soup.find('div', class_='contentbox')

    content_parts = []
    attachments = []
    images = []

    if content_div:
        for child in content_div.find_all(['p', 'table', 'img']):
            if child.name == 'p':
                # 跳过父级div中包含table的<p>（避免表格内容重复）
                p = child.parent
                skip = False
                while p and p != content_div:
                    if p.name == 'div' and p.find('table'):
                        skip = True
                        break
                    p = p.parent
                if skip:
                    continue
                txt = child.get_text(strip=True)
                if txt and len(txt) > 2:
                    content_parts.append(txt)
            elif child.name == 'table':
                html_str = str(child)
                content_parts.append(html_str)
            elif child.name == 'img':
                src = child.get('src', '')
                alt = child.get('alt', '')
                if src:
                    if not src.startswith('http'):
                        src = urljoin(BASE_URL, src)
                    images.append({'src': src, 'alt': alt})
                    content_parts.append('![{0}]({1})'.format(alt, src))

    content = '\n\n'.join(content_parts)

    # 附件检测及降级
    if len(content.strip()) < 20:
        content = '<p><a href="{1}">{0}</a></p>'.format(title, url)
        if images:
            for img in images:
                content += '\n<p><a href="{1}">查看图片</a></p>'.format(img['alt'], img['src'])

    summary = content[:300] if len(content) > 300 else content

    return {
        'title': title,
        'content': content,
        'summary': summary,
        'publish_date': publish_date,
        'source_url': content_source,
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
    }


def main():
    import argparse
    parser = argparse.ArgumentParser(description='当涂县人民政府-公示公告爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES, help='最大爬取页数')
    parser.add_argument('--full', action='store_true', help='全量爬取（44页）')
    args = parser.parse_args()

    max_pages = MAX_PAGES_FULL if args.full else args.max_pages

    session = requests.Session()
    all_items = []
    seen_urls = set()
    total_pages = 0
    empty_pages = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            url = LIST_URL_P1
        else:
            url = LIST_URL_PN.format(page)

        print(f'列表页 {page}/{max_pages}: {url}')
        soup = get_soup(url, session)
        if not soup:
            print(f'  FAILED: 无法获取页面')
            empty_pages += 1
            if empty_pages >= 3:
                print('  连续3页失败，停止')
                break
            continue

        items = extract_list_page(soup)
        if not items:
            print(f'  空列表，停止')
            break

        # 去重
        new_count = 0
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
                new_count += 1

        total_pages += 1
        empty_pages = 0
        print(f'  本页{len(items)}条，新增{new_count}条，累计{len(all_items)}条')

        if new_count == 0:
            print('  无新内容，停止')
            break

        time.sleep(0.5)

    print(f'\n共获取{len(all_items)}篇文章，开始抓取详情...')
    conn = sqlite3.connect(DB_PATH, timeout=60)
    inserted = 0
    updated = 0
    errors = 0

    site_name = '当涂县人民政府'

    for idx, item in enumerate(all_items):
        try:
            print(f'  [{idx+1}/{len(all_items)}] {item["title"][:40]}...')
            soup = get_soup(item['url'], session)
            if not soup:
                errors += 1
                continue

            detail = extract_detail(soup, item['url'])
            if not detail['title']:
                detail['title'] = item['title']

            page_url = item['url']
            publish_date = detail['publish_date'] or item['date']
            content = detail['content']
            summary = detail['summary']
            title = detail['title']
            source_url = detail['source_url']
            attachments = detail['attachments']

            # 库中查重
            cur = conn.execute(
                'SELECT id FROM gov_raw WHERE page_url=? AND site_name=?',
                (page_url, site_name)
            )
            row = cur.fetchone()

            if row:
                conn.execute('''
                    UPDATE gov_raw SET title=?, content=?, summary=?, publish_date=?,
                    source_url=?, attachments=? WHERE id=?
                ''', (title, content, summary, publish_date, source_url, attachments, row[0]))
                updated += 1
            else:
                conn.execute('''
                    INSERT INTO gov_raw (title, content, summary, site_name,
                    page_url, publish_date, source_url, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?)
                ''', (title, content, summary, site_name,
                      page_url, publish_date, source_url, attachments))
                inserted += 1

            time.sleep(0.3)

        except Exception as e:
            print(f'  ERROR: {e}')
            errors += 1

    conn.commit()
    conn.close()
    print(f'\n完成！新增{inserted}条，更新{updated}条，错误{errors}条')


if __name__ == '__main__':
    main()
