#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
砀山县人民政府 - 公示公告爬虫
CMS: Lonsun（蓝汛）
列表: ul.doc_list > li > a[href] + span.date
分页: /content/column/18255837?pageIndex=N（72页，24条/页，1423条）
详情: div.newscontnet + meta ArticleTitle/PubDate/ContentSource
"""

import re
import sys
import json
import time
import requests
import sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://www.dangshan.gov.cn'
LIST_URL_P1 = BASE_URL + '/gsgg/index.html'
LIST_URL_PN = BASE_URL + '/content/column/18255837?pageIndex={}'
DB_PATH = '/root/search.db'
SITE_NAME = '砀山县人民政府-公示公告'
CATEGORY = '安徽'
GROUP = '安徽'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

MAX_PAGES = 5


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def init_db():
    conn = sqlite3.connect(DB_PATH)
    conn.execute('PRAGMA journal_mode=WAL')
    conn.execute('PRAGMA busy_timeout=5000')
    return conn


def get_soup(url, session=None):
    s = session or requests.Session()
    try:
        resp = s.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return None
        return BeautifulSoup(resp.text, 'html.parser')
    except Exception:
        return None


def extract_list_items(soup):
    items = []
    ul = soup.find('ul', class_=lambda x: x and 'doc_list' in str(x))
    if not ul:
        return items
    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        title = (a.get('title') or a.get_text(strip=True) or '').strip()
        if not title:
            continue
        href = a['href'].strip()
        full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
        span = li.find('span')
        date = span.get_text(strip=True) if span else ''
        items.append({'title': title, 'url': full_url, 'date': date})
    return items


def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return None, None, None, None
    except Exception as e:
        print('  [ERROR] fetch detail: %s' % e, file=sys.stderr)
        return None, None, None, None

    soup = BeautifulSoup(r.text, 'html.parser')

    # 标题
    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()

    # 发布日期
    publish_date = ''
    meta_pub = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_pub and meta_pub.get('content'):
        publish_date = meta_pub['content'].strip()[:10]

    # 来源
    source_url = ''
    meta_src = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_src and meta_src.get('content'):
        source_url = meta_src['content'].strip()

    # 正文
    content_div = soup.find('div', class_='newscontnet') or soup.find('div', class_='j-fontContent')
    content_parts = []
    attachments = []

    if content_div:
        for noise in content_div.find_all(['script', 'style']):
            noise.decompose()

        def _has_table_ancestor(el):
            """跳过父级div中包含table的<p>"""
            p = el.parent
            while p and p != content_div:
                if p.name == 'div' and p.find('table'):
                    return True
                p = p.parent
            return False

        for child in content_div.find_all(['p', 'table', 'h1', 'h2', 'h3', 'h4', 'img'], recursive=True):
            if child.name == 'table':
                tbl_html = html_table_to_html(child, url)
                if tbl_html:
                    content_parts.append(tbl_html)
            elif child.name in ('p', 'h1', 'h2', 'h3', 'h4'):
                if _has_table_ancestor(child):
                    continue
                txt = child.get_text(' ', strip=True)
                if txt and len(txt) > 2:
                    content_parts.append(txt)

            elif child.name == 'img':
                src = child.get('src', '')
                alt = child.get('alt', '')
                if src:
                    if not src.startswith('http'):
                        src = urljoin(url, src)
                    content_parts.append('![%s](%s)' % (alt or '', src))

        # 附件提取
        for a in content_div.find_all('a', href=True):
            href = a['href'].strip()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                full_href = href if href.startswith('http') else urljoin(url, href)
                attachments.append({
                    'name': a.get_text(strip=True) or href.split('/')[-1],
                    'url': full_href,
                })

    content = '\n\n'.join(content_parts)

    # 降级：正文太短时嵌入原文链接
    if len(content.strip()) < 20:
        content = '<p><a href="%s">%s</a></p>' % (url, title or url.split('/')[-1])

    summary = content[:200] if len(content) > 200 else content
    attach_json = json.dumps(attachments, ensure_ascii=False) if attachments else ''

    return title, publish_date, source_url, content, summary, attach_json


def main():
    import argparse
    parser = argparse.ArgumentParser(description='砀山县人民政府-公示公告爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES)
    parser.add_argument('--full', action='store_true')
    args = parser.parse_args()

    max_pages = 72 if args.full else args.max_pages

    session = requests.Session()
    all_items = []
    seen_urls = set()

    for page in range(1, max_pages + 1):
        url = LIST_URL_P1 if page == 1 else LIST_URL_PN.format(page)
        print('列表页 %d/%d: %s' % (page, max_pages, url))
        soup = get_soup(url, session)
        if not soup:
            print('  FAILED')
            continue

        items = extract_list_items(soup)
        if not items:
            print('  空列表，停止')
            break

        new_count = 0
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
                new_count += 1

        print('  本页%d条，新增%d条，累计%d条' % (len(items), new_count, len(all_items)))
        if new_count == 0:
            break
        time.sleep(0.5)

    print('\n共%d篇文章，开始抓取详情...' % len(all_items))
    conn = init_db()
    inserted = 0
    updated = 0
    errors = 0

    INSERT_SQL = '''INSERT OR IGNORE INTO gov_raw
        (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
        VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)'''

    for idx, item in enumerate(all_items):
        try:
            print('  [%d/%d] %s...' % (idx + 1, len(all_items), item['title'][:40]))
            title, pub_date, source_url, content, summary, attachments = extract_detail(item['url'])
            if title is None:
                errors += 1
                continue

            final_title = title or item['title']
            final_date = pub_date or item['date']
            final_summary = summary or content[:200] if content else final_title

            conn.execute(INSERT_SQL, (
                SITE_NAME, source_url or item['url'], item['url'],
                final_title.strip(), final_date,
                final_summary.strip(), content,
                CATEGORY, attachments, GROUP,
            ))
            conn.commit()
            inserted += 1
            time.sleep(0.3)

        except Exception as e:
            print('  ERROR: %s' % e)
            errors += 1

    conn.close()
    print('\n完成！新增%d条，更新%d条，错误%d条' % (inserted, updated, errors))
    return inserted


if __name__ == '__main__':
    main()
