#!/usr/bin/env python3
"""
赣县区-环保公示 爬虫
http://www.ganxian.gov.cn/gxqxxgk/hbgss/xxgk_list.shtml
政府信息公开平台，表格+附件+图片处理
"""
import re
import json
import time
import argparse
import requests
from bs4 import BeautifulSoup, Tag

SITE_NAME = "赣县区-环保公示"
GROUP = "生态环境"
BASE_URL = "http://www.ganxian.gov.cn"
LIST_URL = "http://www.ganxian.gov.cn/gxqxxgk/hbgss/xxgk_list.shtml"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
DB_PATH = "/root/search.db"


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def fetch(url, max_retries=3):
    for attempt in range(max_retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            return r.text
        except Exception as e:
            print(f"  Error: {e}, retry {attempt+1}")
        time.sleep(2)
    return None


def parse_list_page(html, page_num):
    """Parse list page and return list of (url, title, date) tuples."""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='pageList') or soup.find('ul', class_='newsList')
    if not ul:
        print(f"  [WARN] No ul.pageList found on page {page_num}")
        return items

    for li in ul.find_all('li', recursive=False):
        # Skip sidebar items (those have nested ul/li structures)
        if li.find('ul'):
            continue
        a = li.find('a')
        time_span = li.find('span', class_='time')
        if not a:
            continue

        href = a.get('href', '')
        if not href:
            continue

        # Full URL
        if not href.startswith('http'):
            href = requests.compat.urljoin(BASE_URL, href)

        # Title - priority: title attribute > a text
        title = a.get('title', '').strip()
        if not title:
            title = a.get_text(strip=True)

        # Date
        date = time_span.get_text(strip=True) if time_span else ""
        # Normalize date
        m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', date)
        if m:
            date = m.group(1).replace('/', '-')

        items.append((href, title, date))

    return items


def parse_detail(html, url):
    """Parse detail page and return (title, content_text, date, attachments_list)."""
    soup = BeautifulSoup(html, 'html.parser')

    # Title
    title_el = soup.find('h1', class_='article-title')
    title = title_el.get_text(strip=True) if title_el else ""

    # Date from meta section - look for date pattern in the meta-main div
    meta_div = soup.find('div', class_='meta-main')
    date = ""
    if meta_div:
        blocks = meta_div.find_all('div', class_='display-block')
        for b in blocks:
            txt = b.get_text(strip=True)
            m = re.match(r'(\d{4}-\d{2}-\d{2})$', txt)
            if m:
                date = m.group(1)
                break

    # Content area
    content_div = soup.find('div', id='zoomcon') or soup.find('div', class_='article-content')
    if not content_div:
        return title, "", date, []

    # Get base path for relative URLs
    base_path = '/'.join(url.split('/')[:-1]) + '/'

    # Extract content, handle tables, images, attachments
    content_parts = []
    attachments = []

    for child in content_div.children:
        if isinstance(child, Tag):
            tag = child.name.lower()
            if tag == 'p':
                # Check if it's a table-wrapping p
                table_inner = child.find('table')
                if table_inner:
                    # Render the table within the paragraph
                    content_parts.append(render_table(table_inner))
                else:
                    # Check if p contains only images
                    imgs = child.find_all('img')
                    if imgs and not child.get_text(strip=True):
                        for img in imgs:
                            process_image(img, base_path, content_parts)
                    else:
                        # Regular paragraph
                        text = body_text(child)
                        if text:
                            content_parts.append(text)

            elif tag == 'table':
                content_parts.append(render_table(child))

            elif tag in ('div', 'section', 'ucapcontent'):
                # Recursively process div content
                for inner in child.children:
                    if isinstance(inner, Tag):
                        if inner.name == 'p':
                            # Check for images in image-only paragraphs
                            imgs = inner.find_all('img')
                            if imgs and not inner.get_text(strip=True):
                                for img in imgs:
                                    process_image(img, base_path, content_parts)
                            else:
                                text = body_text(inner)
                                if text:
                                    content_parts.append(text)
                        elif inner.name == 'table':
                            content_parts.append(render_table(inner))
                        elif inner.name == 'img':
                            process_image(inner, base_path, content_parts)
                        elif inner.name == 'a':
                            process_attachment(inner, base_path, content_parts, attachments)
                # Also get text directly in div
                div_text = body_text(child)
                if div_text:
                    # Only add if we didn't already process children
                    child_parts = [c for c in child.children if isinstance(c, Tag)]
                    if not child_parts:
                        content_parts.append(div_text)

            elif tag == 'img':
                process_image(child, base_path, content_parts)

            elif tag == 'a':
                process_attachment(child, base_path, content_parts, attachments)

    # Also find any standalone attachments in the content area that are not directly children
    for a in content_div.find_all('a', href=True):
        href = a['href']
        if is_attachment_url(href):
            full_url = href if href.startswith('http') else requests.compat.urljoin(base_path, href)
            link_text = a.get_text(strip=True) or full_url.split('/')[-1]
            link_md = f"[{link_text}]({full_url})"
            if link_md not in content_parts and link_md not in [a[1] for a in attachments]:
                attachments.append((link_text, link_md, full_url))
                content_parts.append(link_md)

    # Join content with paragraph separation
    content = '\n\n'.join(content_parts)

    # Clean up extra whitespace in Chinese text
    content = re.sub(r'[ \t]+', ' ', content)
    content = content.strip()

    return title, content, date, attachments


def is_attachment_url(href):
    low = href.lower()
    return any(low.endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar', '.wps', '.ofd'])


def process_image(img_tag, base_path, content_parts):
    src = img_tag.get('src', '')
    if not src:
        return
    full_src = src if src.startswith('http') else requests.compat.urljoin(base_path, src)
    alt = img_tag.get('alt', '')
    if alt:
        content_parts.append(f"![{alt}]({full_src})")
    else:
        content_parts.append(f"![]({full_src})")


def process_attachment(a_tag, base_path, content_parts, attachments):
    href = a_tag.get('href', '')
    if not href:
        return
    if not is_attachment_url(href):
        return
    full_url = href if href.startswith('http') else requests.compat.urljoin(base_path, href)
    link_text = a_tag.get_text(strip=True) or full_url.split('/')[-1]
    link_md = f"[{link_text}]({full_url})"
    attachments.append((link_text, link_md, full_url))
    content_parts.append(link_md)


def render_table(table):
    """Convert HTML table to Markdown pipe table."""
    rows = []
    for tr in table.find_all('tr'):
        cells = []
        for cell in tr.find_all(['td', 'th']):
            text = cell.get_text(strip=True)
            cells.append(text)
        if cells:
            rows.append(cells)

    if not rows:
        return ""

    # Find max columns
    max_cols = max(len(r) for r in rows)
    # Pad rows
    rows = [r + [''] * (max_cols - len(r)) for r in rows]

    lines = []
    # Header
    lines.append('| ' + ' | '.join(rows[0]) + ' |')
    # Separator
    lines.append('| ' + ' | '.join(['---'] * max_cols) + ' |')
    # Data rows
    for row in rows[1:]:
        lines.append('| ' + ' | '.join(row) + ' |')

    return '\n'.join(lines)


def get_total_pages(html):
    """Parse total pages from the pagination script."""
    m = re.search(r"createPageHTML\('page_div',\s*(\d+),\s*\d+,'(\w+)','(\w+)',(\d+)\)", html)
    if m:
        per_page = int(m.group(1))
        total_items = int(m.group(4))
        total_pages = (total_items + per_page - 1) // per_page
        return total_pages, per_page, total_items, m.group(2), m.group(3)
    return 1, 6, 0, 'xxgk_list', 'shtml'


def crawl(max_pages=5, dry_run=False):
    """Main crawl function."""
    print(f"=== {SITE_NAME} 爬虫 ===")
    print(f"目标: {LIST_URL}")
    print(f"最大页数: {max_pages}")
    print()

    all_items = []
    seen_urls = set()

    # Fetch first page to get pagination info
    print(f"[Page 1/??] {LIST_URL}")
    html = fetch(LIST_URL)
    if not html:
        print("  [FAIL] Failed to fetch list page")
        return all_items

    total_pages, per_page, total_items, prefix, ext = get_total_pages(html)
    print(f"  每页 {per_page} 条, 共 {total_items} 条, {total_pages} 页")

    # Parse first page
    items = parse_list_page(html, 1)
    print(f"  解析到 {len(items)} 条")
    for item in items:
        if item[0] not in seen_urls:
            seen_urls.add(item[0])
            all_items.append(item)

    # Fetch remaining pages (up to max_pages)
    pages_to_fetch = min(max_pages, total_pages)
    for page in range(2, pages_to_fetch + 1):
        if prefix == 'xxgk_list':
            if page == 2:
                page_url = f"{BASE_URL}/gxqxxgk/hbgss/{prefix}_2.shtml"
            else:
                page_url = f"{BASE_URL}/gxqxxgk/hbgss/{prefix}_{page}.shtml"
        else:
            page_url = f"{BASE_URL}/gxqxxgk/hbgss/{prefix}_{page}.{ext}"

        print(f"[Page {page}/{pages_to_fetch}] {page_url}")
        html = fetch(page_url)
        if not html:
            print(f"  [WARN] Failed to fetch page {page}")
            continue

        items = parse_list_page(html, page)
        print(f"  解析到 {len(items)} 条")
        for item in items:
            if item[0] not in seen_urls:
                seen_urls.add(item[0])
                all_items.append(item)

    print(f"\n总计: {len(all_items)} 条待处理")

    if dry_run:
        print("\n=== DRY RUN - 仅列出 ===")
        for url, title, date in all_items:
            print(f"  {date} | {title[:60]}... | {url}")
        return all_items

    # Fetch detail pages
    print("\n=== 获取详情 ===")
    import sqlite3

    for i, (url, list_title, list_date) in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {url.split('/')[-1][:20]}... {list_title[:40]}...")
        html = fetch(url)
        if not html:
            print(f"    [SKIP] Failed to fetch detail")
            continue

        title, content, date, attachments = parse_detail(html, url)

        # Use title from detail page if available, fall back to list title
        final_title = title or list_title
        final_date = date or list_date

        # Build attachments text
        attachment_links = []
        for name, md_link, full_url in attachments:
            attachment_links.append(md_link)

        # Embed attachments into content
        full_content = content
        if attachment_links:
            if full_content.strip():
                full_content += "\n\n**附件:**\n" + "\n".join(attachment_links)
            else:
                full_content = "**附件:**\n" + "\n".join(attachment_links)

        # If content is empty but there are attachments, use just the attachment links
        if not full_content.strip() and attachment_links:
            full_content = "\n".join(attachment_links)

        # If still empty, try to get the raw text to see if it's image-only
        if not full_content.strip():
            # Check if the page has any images even if get_text was empty
            content_div = BeautifulSoup(html, 'html.parser').find('div', id='zoomcon') or BeautifulSoup(html, 'html.parser').find('div', class_='article-content')
            if content_div and content_div.find('img'):
                # Image-only content - extract image links
                base_path = '/'.join(url.split('/')[:-1]) + '/'
                img_parts = []
                for img in content_div.find_all('img'):
                    src = img.get('src', '')
                    if src:
                        full_src = src if src.startswith('http') else requests.compat.urljoin(base_path, src)
                        img_parts.append(f"![]({full_src})")
                if img_parts:
                    full_content = "\n".join(img_parts)

        if not full_content.strip():
            print(f"    [SKIP] Empty content")
            continue

        # Save to DB
        try:
            conn = sqlite3.connect(DB_PATH, timeout=60)
            c = conn.cursor()
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (page_url, title, publish_date, content, summary, site_name, group_name) VALUES (?, ?, ?, ?, ?, ?, ?)",
                (url, final_title, final_date, full_content, final_title, SITE_NAME, GROUP)
            )
            new = c.rowcount
            conn.commit()
            conn.close()
            print(f"    {'[NEW]' if new else '[skip]'} {final_title[:50]}... | {final_date}")
        except Exception as e:
            print(f"    [DB ERROR] {e}")

    print(f"\n=== 完成 ===")
    return all_items


if __name__ == '__main__':
    parser = argparse.ArgumentParser(description=f'{SITE_NAME} 爬虫')
    parser.add_argument('--max-pages', type=int, default=5, help='最大爬取页数 (default: 5)')
    parser.add_argument('--dry-run', action='store_true', help='仅列出不入库')
    parser.add_argument('--pages', type=int, help='别名 (--max-pages 的简写)')
    args = parser.parse_args()

    if args.pages is not None:
        args.max_pages = args.pages

    crawl(max_pages=args.max_pages, dry_run=args.dry_run)
