#!/usr/bin/env python3
"""郯城县-公示公告爬虫

VSB 9 (Visual SiteBuilder 博达)
列表: /zfxxgkpt/xzbmxxgk/xxgkml/gsgg.htm (第1页, 50条/页)
      第N页: gsgg/{id}.htm (递减ID)
详情: /info/1694/XXXXX.htm
标题: meta ArticleTitle / pageTitle
日期: meta PubDate
正文: div.newscontent_s > p, table

Usage:
  python3 crawl_tancheng_gsgg.py --pages 5    # 前5页
  python3 crawl_tancheng_gsgg.py --pages 1    # 日跑增量
"""

import requests
import re
import sys
import json
import time
import os
import argparse
from bs4 import BeautifulSoup, Tag, NavigableString
from urllib.parse import urljoin

sys.path.insert(0, '/root/gov_crawler')
from crawler_lib import push_to_searchdb
import sqlite3
import urllib.parse

DB_PATH = '/root/search.db'


def get_db_connection():
    try:
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("PRAGMA busy_timeout=30000")
        conn.execute("PRAGMA journal_mode=WAL")
        return conn
    except Exception as e:
        print(f'  WARN DB connect: {e}')
        return None


BASE_URL = 'http://www.tancheng.gov.cn'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

SITE_NAME = '郯城县-公示公告'
LIST_PATH = 'zfxxgkpt/xzbmxxgk/xxgkml'

sess = requests.Session()
sess.headers.update(HEADERS)


def fetch_page(url):
    resp = sess.get(url, timeout=15)
    resp.encoding = 'utf-8'
    return resp.text


def parse_list(html):
    """解析列表页"""
    items = []
    # <li id="line247500_N">
    #   <a class="bt" href="../../../info/1694/XXXXX.htm">标题（有省略号...）</a>
    #   <span class="time">2026-07-17</span>
    # </li>
    pattern = re.compile(
        r'<li[^>]*>.*?'
        r'<a[^>]*href="([^"]+)"[^>]*class="bt"[^>]*>([^<]+)</a>\s*'
        r'<span[^>]*class="time">(\d{4}-\d{2}-\d{2})</span>',
        re.DOTALL
    )
    for m in pattern.finditer(html):
        href = m.group(1).strip()
        title = m.group(2).strip()
        pub_date = m.group(3).strip()

        if not title or not href:
            continue
        if '/info/' not in href:
            continue

        # Resolve: href is like ../../../info/1694/337546.htm
        if href.startswith('../../../'):
            url = BASE_URL + href[8:]  # removes ../../../
        elif href.startswith('../../'):
            url = BASE_URL + href[5:]
        elif href.startswith('../'):
            url = BASE_URL + href[2:]
        elif not href.startswith('http'):
            url = BASE_URL + '/' + href.lstrip('/')
        else:
            url = href

        items.append({'title': title, 'url': url, 'pub_date': pub_date})
    return items


def get_next_page_url(html, current_url):
    """从分页区域获取"下页"链接"""
    soup = BeautifulSoup(html, 'html.parser')
    page_div = soup.find('span', class_='p_pages')
    if not page_div:
        return None
    for a in page_div.find_all('a', href=True):
        txt = a.get_text(strip=True)
        if '下页' in txt or '下一页' in txt:
            href = a['href']
            if 'javascript' in href:
                return None
            if href.startswith('/'):
                return BASE_URL + href
            if href.startswith('../'):
                base_dir = current_url.rsplit('/', 1)[0]
                parts = base_dir.split('/')
                count = 0
                while href.startswith('../'):
                    href = href[3:]
                    count += 1
                if count >= len(parts):
                    return BASE_URL + '/' + href
                resolved = '/'.join(parts[:-count]) + '/' + href
                return resolved
            base_dir = current_url.rsplit('/', 1)[0]
            return base_dir + '/' + href
    return None


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def _extract_p_text(p_tag, base_url):
    """从 <p> 标签中提取纯文本"""
    if p_tag.find(['table']):
        return ''
    txt = ''
    for child in p_tag.children:
        if isinstance(child, NavigableString):
            t = str(child).strip()
            if t:
                txt += t + ' '
        elif child.name == 'br':
            txt += '\n'
        elif child.name == 'img':
            src = child.get('src', '')
            alt = child.get('alt', '')
            if src:
                full_src = src if src.startswith('http') else urljoin(base_url, src)
                txt += f'![{alt}]({full_src}) '
        elif child.name == 'a':
            a_href = child.get('href', '')
            a_text = child.get_text(strip=True)
            if a_href and a_text:
                full_href = a_href if a_href.startswith('http') else urljoin(base_url, a_href)
                txt += f'[{a_text}]({full_href}) '
        elif child.name in ('span', 'strong', 'b', 'em', 'u', 'font'):
            txt += child.get_text(' ', strip=True) + ' '
    txt = re.sub(r'[ \t]+', ' ', txt).strip()
    txt = txt.replace('\xa0', ' ')
    txt = re.sub(r'[ \t]+', ' ', txt).strip()
    return txt


def fetch_detail(item):
    """抓取详情页"""
    url = item['url']
    try:
        html = fetch_page(url)
    except Exception as e:
        print(f'  WARN skip: {url} - {e}')
        return item['title'], item.get('pub_date', ''), '', []

    soup = BeautifulSoup(html, 'html.parser')

    # 标题 - 优先 meta pageTitle / ArticleTitle
    title = ''
    for name in ['ArticleTitle', 'pageTitle']:
        mt = soup.find('meta', attrs={'name': name})
        if mt and mt.get('content'):
            title = mt['content'].strip()
            break
    if not title:
        h3 = soup.find('h3')
        if h3:
            title = h3.get_text(strip=True)
    if not title:
        mt = soup.find('meta', attrs={'name': re.compile('pageTitle', re.I)})
        if mt and mt.get('content'):
            title = mt['content'].strip()
    if not title:
        title = item['title']

    # 日期
    pub_date = item.get('pub_date', '')
    md = soup.find('meta', attrs={'name': 'PubDate'})
    if md and md.get('content'):
        m = re.match(r'(\d{4}-\d{2}-\d{2})', md['content'].strip())
        if m:
            pub_date = m.group(1)

    # 正文 - div.newscontent_s (VSB)
    content_view = soup.find('div', class_='newscontent_s')
    content = ''
    attachments = []

    if content_view:
        parts = []
        seen_tables = set()

        for elem in content_view.find_all(['p', 'table', 'img'], recursive=True):
            style = elem.get('style', '')
            if 'display:none' in style.replace(' ', ''):
                continue
            if elem.name == 'p' and elem.find_parent('table'):
                continue
            if elem.name == 'table':
                table_str = str(elem)[:100]
                if table_str in seen_tables:
                    continue
                seen_tables.add(table_str)
                md_table = html_table_to_html(elem)
                if md_table:
                    parts.append(md_table)
            elif elem.name == 'p':
                txt = _extract_p_text(elem, url)
                if txt:
                    parts.append(txt)
            elif elem.name == 'img':
                src = elem.get('src', '')
                alt = elem.get('alt', '')
                if src:
                    full_src = src if src.startswith('http') else urljoin(url, src)
                    parts.append(f'![{alt}]({full_src})')

        content = '\n\n'.join(parts)

        # 附件
        for a in content_view.find_all('a', href=True):
            href = a.get('href', '')
            text = a.get_text(strip=True)
            if href and text and re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                if not href.startswith('http'):
                    href = urljoin(url, href)
                attachments.append({'name': text, 'url': href})
                content += f'\n\n附件：<p><a href="{href}">{text}</a></p>'

    return title, pub_date, content, attachments


def main():
    parser = argparse.ArgumentParser(description='郯城县-公示公告')
    parser.add_argument('--pages', type=int, default=0, help='页数，0=全量')
    args = parser.parse_args()

    max_pages = args.pages if args.pages > 0 else 5
    all_items = []

    list_url = f'{BASE_URL}/{LIST_PATH}/gsgg.htm'
    current_url = list_url

    for page in range(1, max_pages + 1):
        print(f'Page {page}/{max_pages}...')
        try:
            html = fetch_page(current_url)
            items = parse_list(html)
            if not items:
                print(f'  empty, stop')
                break
            print(f'  found {len(items)} items')
            all_items.extend(items)
        except Exception as e:
            print(f'  WARN: {e}')
            break

        next_url = get_next_page_url(html, current_url)
        if not next_url:
            print(f'  no next page, stop')
            break
        current_url = next_url
        time.sleep(0.3)

    print(f'\nTotal list: {len(all_items)} items')

    if not all_items:
        print('No items found, abort')
        return

    records = []
    for i, item in enumerate(all_items):
        print(f'  [{i+1}/{len(all_items)}] {item["title"][:40]}...')
        title, pub_date, content, attachments = fetch_detail(item)
        if not title:
            title = item['title']
        title = re.sub(r'\s+', ' ', title).strip()
        summary = content[:500] if content else ''

        record = {
            'url': item['url'],
            'source_url': item['url'],
            'title': title,
            'content': content,
            'summary': summary,
            'pub_date': pub_date,
            'site_name': SITE_NAME,
            'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
        }
        records.append(record)
        time.sleep(0.2)

    print(f'\nInserting {len(records)} records...')
    if records:
        push_to_searchdb(records, SITE_NAME)

    conn = get_db_connection()
    if conn:
        try:
            cursor = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
            raw_count = cursor.fetchone()[0]
            cursor = conn.execute("SELECT COUNT(*) FROM gov_search WHERE site_name=?", (SITE_NAME,))
            fts_count = cursor.fetchone()[0]
            print(f'DB: {raw_count} rows (FTS: {fts_count} rows)')
        except Exception as e:
            print(f'  WARN DB query: {e}')
        finally:
            conn.close()

    print('DONE')


if __name__ == '__main__':
    main()
