#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
顺昌县人民政府 - 公示公告爬虫
CMS: 自定义（非Lonsun）
列表: div.mid-mj-list > ul#resources > li > span.list-content > a[href] + span.list-time
分页: /cms/sitemanage/index.shtml?siteId=30128195135920000&page=N（756页，20条/页）
详情: div.content + meta ArticleTitle/PubDate
"""

import re
import sys
import json
import time
import requests
import sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'http://www.fjsc.gov.cn'
LIST_URL_P1 = BASE_URL + '/cms/html/scxrmzf/gsgg/index.html'
LIST_URL_PN = BASE_URL + '/cms/sitemanage/index.shtml?siteId=30128195135920000&page={}'
DB_PATH = '/root/search.db'
SITE_NAME = '顺昌县人民政府-公示公告'
CATEGORY = '福建'
GROUP = '福建'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

MAX_PAGES = 5


def normalize_date(d):
    """页面 meta PubDate 用 em-dash(—) 而列表页用连字符(-)，统一为半角连字符"""
    if not d:
        return d
    return d.replace('—', '-').replace('–', '-').replace('－', '-')


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute('PRAGMA journal_mode=WAL')
    conn.execute('PRAGMA busy_timeout=5000')
    return conn


def get_soup(url, session=None):
    s = session or requests.Session()
    try:
        resp = s.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return None
        return BeautifulSoup(resp.text, 'html.parser')
    except Exception:
        return None


def extract_list_items(soup):
    items = []
    div = soup.find('div', class_='mid-mj-list')
    if not div:
        return items
    ul = div.find('ul', id='resources')
    if not ul:
        # 直接找 li
        for li in div.find_all('li'):
            a = li.find('a')
            span_time = li.find('span', class_='list-time')
            if a and a.get('href'):
                title = a.get('title') or a.get_text(strip=True) or ''
                title = title.strip()
                if not title:
                    continue
                href = a['href'].strip()
                full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
                date = normalize_date(span_time.get_text(strip=True)) if span_time else ''
                items.append({'title': title, 'url': full_url, 'date': date})
        return items

    for li in ul.find_all('li', recursive=False):
        a = li.find('a')
        span_time = li.find('span', class_='list-time')
        if a and a.get('href'):
            title = a.get('title') or a.get_text(strip=True) or ''
            title = title.strip()
            if not title:
                continue
            href = a['href'].strip()
            full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
            date = normalize_date(span_time.get_text(strip=True)) if span_time else ''
            items.append({'title': title, 'url': full_url, 'date': date})

    # 如果 ul#resources 没找到，fallback到div下所有a
    if not items:
        for a in div.find_all('a'):
            href = a.get('href', '')
            title = a.get_text(strip=True)
            if href and title and len(title) > 5:
                full_url = href if href.startswith('http') else urljoin(BASE_URL, href)
                items.append({'title': title, 'url': full_url, 'date': ''})

    return items


def extract_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return None, None, None, None, None, None
    except Exception as e:
        print('  [ERROR] fetch detail: %s' % e, file=sys.stderr)
        return None, None, None, None, None, None

    soup = BeautifulSoup(r.text, 'html.parser')

    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)

    publish_date = ''
    meta_pub = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_pub and meta_pub.get('content'):
        publish_date = normalize_date(meta_pub['content'].strip()[:10])

    source_url = ''
    meta_src = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_src and meta_src.get('content'):
        source_url = meta_src['content'].strip()

    content_div = soup.find('div', class_='content') or soup.find('div', class_='TRS_Editor')
    content_parts = []
    attachments = []

    if content_div:
        for noise in content_div.find_all(['script', 'style']):
            noise.decompose()

        def _has_table_ancestor(el):
            p = el.parent
            while p and p != content_div:
                if p.name == 'div' and p.find('table'):
                    return True
                p = p.parent
            return False

        for child in content_div.find_all(['p', 'table', 'img'], recursive=True):
            if child.name == 'table':
                tbl_html = html_table_to_html(child, url)
                if tbl_html:
                    content_parts.append(tbl_html)
            elif child.name == 'p':
                if _has_table_ancestor(child):
                    continue
                txt = child.get_text(' ', strip=True)
                if txt and len(txt) > 2:
                    content_parts.append(txt)

            elif child.name == 'img':
                src = child.get('src', '')
                alt = child.get('alt', '')
                if src:
                    if not src.startswith('http'):
                        src = urljoin(url, src)
                    content_parts.append('<p><a href="%s">查看图片</a></p>' % (src,))

        # 附件
        for a in content_div.find_all('a', href=True):
            href = a['href'].strip()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                full_href = href if href.startswith('http') else urljoin(url, href)
                attachments.append({
                    'name': a.get_text(strip=True) or href.split('/')[-1],
                    'url': full_href,
                })

    content = '\n\n'.join(content_parts)

    if len(content.strip()) < 20:
        content = '<p><a href="%s">%s</a></p>' % (url, title or url.split('/')[-1])

    summary = content[:200] if len(content) > 200 else content
    attach_json = json.dumps(attachments, ensure_ascii=False) if attachments else ''

    return title, publish_date, source_url, content, summary, attach_json


def main():
    import argparse
    parser = argparse.ArgumentParser(description='顺昌县人民政府-公示公告爬虫')
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES)
    parser.add_argument('--full', action='store_true')
    args = parser.parse_args()

    max_pages = 756 if args.full else args.max_pages

    session = requests.Session()
    all_items = []
    seen_urls = set()

    for page in range(1, max_pages + 1):
        url = LIST_URL_P1 if page == 1 else LIST_URL_PN.format(page)
        print('列表页 %d/%d: %s' % (page, max_pages, url))
        soup = get_soup(url, session)
        if not soup:
            print('  FAILED')
            continue

        items = extract_list_items(soup)
        if not items:
            print('  空列表，停止')
            break

        new_count = 0
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
                new_count += 1

        print('  本页%d条，新增%d条，累计%d条' % (len(items), new_count, len(all_items)))
        if new_count == 0:
            break
        time.sleep(0.5)

    print('\n共%d篇文章，开始抓取详情...' % len(all_items))
    conn = init_db()
    inserted = 0
    errors = 0

    INSERT_SQL = '''INSERT OR IGNORE INTO gov_raw
        (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
        VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)'''

    for idx, item in enumerate(all_items):
        try:
            print('  [%d/%d] %s...' % (idx + 1, len(all_items), item['title'][:40]))
            title, pub_date, src_url, content, summary, attachments = extract_detail(item['url'])
            if title is None:
                errors += 1
                continue

            final_title = title or item['title']
            final_date = pub_date or item['date']
            final_summary = summary or content[:200] if content else final_title

            conn.execute(INSERT_SQL, (
                SITE_NAME, src_url or item['url'], item['url'],
                final_title.strip(), final_date,
                final_summary.strip(), content,
                CATEGORY, attachments, GROUP,
            ))
            conn.commit()
            inserted += 1
            time.sleep(0.3)

        except Exception as e:
            print('  ERROR: %s' % e)
            errors += 1

    conn.close()
    print('\n完成！新增%d条，错误%d条' % (inserted, errors))
    return inserted


if __name__ == '__main__':
    main()
