#!/usr/bin/env python3
"""
东营港经济开发区 - 环境管理 (TRS WCM JPage)
http://www.dypedz.gov.cn/col/col229138/index.html
正文缺失修复：递归遍历 wenzhang_nr 内所有子节点，包括 div/p/table/img/a
"""
import requests
import re
import sys
import os
import json
from bs4 import BeautifulSoup, Tag, NavigableString
import bs4

BASE_URL = "http://www.dypedz.gov.cn"
LIST_URL = "http://www.dypedz.gov.cn/col/col229138/index.html"
SITE_NAME = "东营港经济开发区"
COLUMN_NAME = "环境管理"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": LIST_URL,
}

sys.path.insert(0, "/root/gov_crawler")
from crawler_lib import push_to_searchdb

session = requests.Session()
session.headers.update(HEADERS)


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def extract_list_page(url):
    """Extract article list from embedded XML in the main page"""
    r = session.get(url, timeout=30)
    r.encoding = 'utf-8'
    html = r.text
    articles = []
    pattern = r'<record><!\[CDATA\[.*?<a href="(https?://[^"]+)"\s+title="([^"]*)"[^>]*>(.*?)</a>.*?<span>([^<]+)</span>'
    matches = re.findall(pattern, html, re.DOTALL)
    for url, title, link_text, date_str in matches:
        title = title.strip()
        if not title:
            title = re.sub(r'<[^>]+>', '', link_text).strip()
        # Strip site/column prefix
        prefix = f"{SITE_NAME} {COLUMN_NAME} "
        if title.startswith(prefix):
            title = title[len(prefix):]
        articles.append({'url': url, 'title': title, 'date': date_str.strip()})
    return articles


def parse_detail(url):
    """Extract full content from detail page - recursive div handling"""
    r = session.get(url, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')

    # Title from wenzhang_tit
    title_div = soup.find('div', class_='wenzhang_tit')
    title = ''
    if title_div:
        title = title_div.get_text(strip=True)
        title = re.sub(r'<!--.*?-->', '', title).strip()
        prefix = f"{SITE_NAME} {COLUMN_NAME} "
        if title.startswith(prefix):
            title = title[len(prefix):]

    # Fallback to meta ArticleTitle
    if not title:
        meta = soup.find('meta', attrs={'name': 'ArticleTitle'})
        if meta:
            title = meta.get('content', '')

    # Date from meta (TRS always has pubdate meta)
    pub_date = ''
    meta_date = soup.find('meta', attrs={'name': 'pubdate'})
    if meta_date:
        pub_date = meta_date.get('content', '')

    # Fallback to visible date td
    if not pub_date:
        date_td = soup.find('td', class_='riqi')
        if date_td:
            text = date_td.get_text(strip=True)
            m = re.search(r'(\d{4}-\d{2}-\d{2}\s*\d{2}:\d{2}:\d{2})', text)
            if m:
                pub_date = m.group(1)

    # Source
    source = ''
    source_meta = soup.find('meta', attrs={'name': 'contentSource'})
    if source_meta:
        source = source_meta.get('content', '')
    if not source:
        laiyuan_td = soup.find('td', class_='laiyuan')
        if laiyuan_td:
            text = laiyuan_td.get_text(strip=True)
            text = re.sub(r'^信息来源[：:]', '', text).strip()
            source = text

    # Body from wenzhang_nr
    body_parts = []
    nr_div = soup.find('div', class_='wenzhang_nr')
    if nr_div:
        body_parts = extract_node_content(nr_div)

    body = '\n\n'.join(body_parts)
    return title, pub_date, source, body


def extract_node_content(node):
    """Recursively extract content from a node, handling all tag types"""
    parts = []
    for child in node.children:
        if isinstance(child, NavigableString):
            text = child.strip()
            # Skip TRS/ZJEG comment markers and HTML comment remnants
            if text and text != '\n' and not re.match(r'^<|\$|ZJEG|ContentStart|ContentEnd', text):
                parts.append(text)
        elif isinstance(child, bs4.Comment):
            # Skip HTML comments
            pass
        elif isinstance(child, Tag):
            tag = child.name.lower()
            if tag in ('script', 'style', 'noscript'):
                continue
            elif tag == 'meta':
                # BS sometimes treats meta.ContentStart as a container tag
                # Recurse into it to get the real content
                sub_parts = extract_node_content(child)
                parts.extend(sub_parts)
            elif tag == 'p':
                text = body_text(child)
                # Check if p contains attachment links (a + img icon)
                links = child.find_all('a')
                has_attachment = any(a.find('img') for a in links) if links else False

                if has_attachment:
                    # Attachment paragraph - extract links as Markdown
                    for a in child.find_all('a'):
                        href = a.get('href', '')
                        if href and not href.startswith('http'):
                            href = BASE_URL + href
                        a_text = a.get_text(strip=True)
                        if href and a_text:
                            parts.append(f"[{a_text}]({href})")
                        elif href:
                            parts.append(f"[{os.path.basename(href)}]({href})")
                elif text:
                    parts.append(text)
            elif tag == 'div':
                # Check if contains a table
                table = child.find('table')
                if table:
                    md_table = table_to_markdown(table)
                    if md_table:
                        parts.append(md_table)
                else:
                    # Recurse into div
                    sub_parts = extract_node_content(child)
                    parts.extend(sub_parts)
            elif tag == 'table':
                md_table = table_to_markdown(child)
                if md_table:
                    parts.append(md_table)
            elif tag == 'img':
                src = child.get('src', '')
                alt = child.get('alt', '')
                if src and not src.startswith('http'):
                    src = BASE_URL + src
                if src:
                    parts.append(f"![{alt}]({src})")
            elif tag == 'a':
                href = child.get('href', '')
                if href and not href.startswith('http'):
                    href = BASE_URL + href
                a_text = child.get_text(strip=True)
                if href and a_text:
                    parts.append(f"[{a_text}]({href})")
                elif href:
                    parts.append(f"[{href}]({href})")
            elif tag == 'br':
                pass  # Line breaks handled by get_text
            else:
                # Other tags (span, strong, em, etc.) - extract text
                text = child.get_text(strip=True)
                if text:
                    parts.append(text)
    return parts


def table_to_markdown(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def deduplicate_attachments(body):
    """Remove duplicate attachment references"""
    lines = body.split('\n\n')
    seen = set()
    deduped = []
    for line in lines:
        key = line.strip()
        if key in seen:
            continue
        seen.add(key)
        deduped.append(line)
    return '\n\n'.join(deduped)


def main():
    max_pages = 5
    if len(sys.argv) > 1:
        for arg in sys.argv[1:]:
            if arg.startswith('--max-pages='):
                max_pages = int(arg.split('=')[1])

    all_articles = []
    print("[*] 获取列表第1页...")
    articles = extract_list_page(LIST_URL)
    print(f"  -> 获取 {len(articles)} 条")
    all_articles.extend(articles)

    # Pages 2+: try JPage dataproxy
    for page in range(2, max_pages + 1):
        proxy_url = f"{BASE_URL}/module/web/jpage/dataproxy.jsp?page={page}&col=1&webid=277&path={BASE_URL}/&columnid=229138&unitid=669879&webname=%E4%B8%9C%E8%90%A5%E6%B8%AF%E7%BB%8F%E6%B5%8E%E5%BC%80%E5%8F%91%E5%8C%BA&permissiontype=0"
        try:
            r = session.get(proxy_url, timeout=15)
            content = r.text.strip()
            if not content or len(content) < 50:
                print(f"  -> 第{page}页无数据")
                break
            pattern = r'<record><!\[CDATA\[.*?<a href="(https?://[^"]+)"\s+title="([^"]*)"[^>]*>(.*?)</a>.*?<span>([^<]+)</span>'
            matches = re.findall(pattern, content, re.DOTALL)
            if not matches:
                print(f"  -> 第{page}页无更多数据")
                break
            for url, title, link_text, date_str in matches:
                title = title.strip()
                if not title:
                    title = re.sub(r'<[^>]+>', '', link_text).strip()
                prefix = f"{SITE_NAME} {COLUMN_NAME} "
                if title.startswith(prefix):
                    title = title[len(prefix):]
                all_articles.append({'url': url, 'title': title, 'date': date_str.strip()})
            print(f"  -> 第{page}页: {len(matches)} 条")
        except Exception as e:
            print(f"  [!] 第{page}页请求失败: {e}")
            break

    print(f"\n[*] 共获取 {len(all_articles)} 篇文章")

    db_items = []
    for idx, article in enumerate(all_articles, 1):
        url = article['url']
        list_title = article['title']
        list_date = article['date']
        print(f"\n[{idx}/{len(all_articles)}] {list_title}")
        try:
            detail_title, detail_date, source, body = parse_detail(url)
            final_title = list_title
            final_date = detail_date or list_date
            body = deduplicate_attachments(body)

            # Summary: first 200 chars without whitespace
            summary = re.sub(r'\s+', '', body)[:200] if body else ''

            # Attachments found in body
            attachments = []
            for line in body.split('\n\n'):
                if line.startswith('[') and '](' in line and line.endswith(')'):
                    attachments.append(line)

            db_items.append({
                'site_name': SITE_NAME,
                'source_url': url,
                'url': url,
                'title': final_title,
                'pub_date': final_date,
                'summary': summary,
                'content': body or '(无正文内容)',
                'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
            })

            # Print preview of first article
            if idx == 1:
                print(f"  ├ 正文预览: {body[:200]}...")

        except Exception as e:
            print(f"  !! 错误: {e}")

    # Push to DB
    if db_items:
        print(f"\n[*] 入库 {len(db_items)} 条...")
        push_to_searchdb(db_items, SITE_NAME)
        print(f"✅ 入库完成")
    else:
        print("No items to push")

    print(f"\n{'='*50}")
    print(f"总计: {len(all_articles)} 条")


if __name__ == '__main__':
    main()
