#!/usr/bin/env divython3
"""
Crawl 盘锦辽滨沿海经济技术开发区 - 环境保护
https://ldwxq.panjin.gov.cn/13731/
Pagination: list-2.html ... list-30.html (30 pages, 449 items)
List: <a href="..." title="完整标题"><span>标题</span><span>日期</span></a>
Detail: <div class="Q25O_OpennavRarbox">content HTML</div>
"""

import requests, json, re, os, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://ldwxq.panjin.gov.cn/13731/'
OUTPUT_FILE = '/root/gov_crawler/output/panjin_hjbh.jsonl'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9,en;q=0.8',
}

os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)

session = requests.Session()
session.headers.update(HEADERS)

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def get_page_url(page_num):
    if page_num == 1:
        return BASE_URL
    return "https://ldwxq.panjin.gov.cn/13731/list-" + str(page_num) + ".html"

def parse_list(html):
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for a in soup.find_all('a', href=True, title=True):
        href = a['href']
        title = a['title'].strip()
        if not title:
            continue
        if 'content-' not in href and '/202' not in href:
            continue
        if not href.startswith('http'):
            href = urljoin(BASE_URL, href)
        date = ''
        for span in a.find_all('span'):
            txt = span.get_text(strip=True)
            m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', txt)
            if m:
                date = m.group(1)
                break
        items.append({'url': href, 'title': title, 'date': date})
    return items

def parse_detail(url, html):
    soup = BeautifulSoup(html, 'html.parser')
    title_el = soup.find('div', class_='Q25O_OpennavRartitle')
    list_title = title_el.get_text(strip=True) if title_el else ''
    date_el = soup.find('div', class_='Q25O_OpennavRartitledate')
    pub_date = ''
    if date_el:
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', date_el.get_text(strip=True))
        if m:
            pub_date = m.group(1)
    content_div = soup.find('div', class_='Q25O_OpennavRarbox')
    content_html = ''
    summary = ''
    attachments = []
    if content_div:
        for tag in content_div.find_all(True):
            attrs_to_keep = ['href', 'src', 'alt', 'target']
            for attr in list(tag.attrs):
                if attr not in attrs_to_keep:
                    del tag[attr]
        content_html = str(content_div)
        summary = body_text(content_div)
        for a in content_div.find_all('a', href=True):
            ahref = a.get('href', '')
            if any(ahref.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                full_url = ahref if ahref.startswith('http') else urljoin(url, ahref)
                attachments.append({'url': full_url, 'text': a.get_text(strip=True) or os.path.basename(ahref)})
    return {
        'title': list_title,
        'pub_date': pub_date,
        'content_html': content_html,
        'summary': summary[:500] if summary else '',
        'attachments': attachments
    }

def get_total_pages(html):
    m = re.search(r'\d+/(\d+)页', html)
    if m:
        return int(m.group(1))
    max_page = 1
    for m in re.finditer(r'list-(\d+)', html):
        pn = int(m.group(1))
        if pn > max_page:
            max_page = pn
    return max_page

def crawl_all():
    print('Fetching page 1...')
    resp = session.get(BASE_URL, timeout=30)
    resp.encoding = 'utf-8'
    total_pages = get_total_pages(resp.text)
    print('Total pages:', total_pages)
    all_items = []
    for page in range(1, total_pages + 1):
        url = get_page_url(page)
        print('  Page', page, '/', total_pages)
        try:
            if page == 1:
                resp_html = resp.text
            else:
                resp = session.get(url, timeout=30)
                resp.encoding = 'utf-8'
                resp_html = resp.text
            items = parse_list(resp_html)
            print('    Found', len(items), 'items')
            all_items.extend(items)
        except Exception as e:
            print('    [ERROR]', e)
        time.sleep(0.5)
    seen_urls = set()
    unique_items = []
    for item in all_items:
        if item['url'] not in seen_urls:
            seen_urls.add(item['url'])
            unique_items.append(item)
    print('\nTotal unique items:', len(unique_items))
    record_count = 0
    with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:
        for i, item in enumerate(unique_items):
            print('  [' + str(i+1) + '/' + str(len(unique_items)) + ']', item['title'][:40]+'...')
            try:
                resp = session.get(item['url'], timeout=30)
                resp.encoding = 'utf-8'
                detail = parse_detail(item['url'], resp.text)
                title = detail['title'] or item['title']
                record = {
                    'title': title,
                    'url': item['url'],
                    'date': detail['pub_date'] or item['date'],
                    'content': detail['content_html'],
                    'summary': detail['summary'],
                    'site_name': '盘锦辽滨沿海经济技术开发区',
                    'group': '盘锦辽滨',
                    'attachments': json.dumps(detail['attachments'], ensure_ascii=False) if detail['attachments'] else '',
                }
                f.write(json.dumps(record, ensure_ascii=False) + '\n')
                record_count += 1
            except Exception as e:
                print('    [ERROR]', item['url'], ':', e)
            time.sleep(0.3)
    print('\nDone!', record_count, 'records written to', OUTPUT_FILE)

if __name__ == '__main__':
    if '--incremental' in sys.argv:
        print('Incremental mode')
        resp = session.get(BASE_URL, timeout=30)
        resp.encoding = 'utf-8'
        items = parse_list(resp.text)
        print('Found', len(items), 'items on page 1')
        with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:
            for item in items:
                try:
                    resp = session.get(item['url'], timeout=30)
                    resp.encoding = 'utf-8'
                    detail = parse_detail(item['url'], resp.text)
                    title = detail['title'] or item['title']
                    record = {
                        'title': title,
                        'url': item['url'],
                        'date': detail['pub_date'] or item['date'],
                        'content': detail['content_html'],
                        'summary': detail['summary'],
                        'site_name': '盘锦辽滨沿海经济技术开发区',
                        'group': '盘锦辽滨',
                        'attachments': json.dumps(detail['attachments'], ensure_ascii=False) if detail['attachments'] else '',
                    }
                    f.write(json.dumps(record, ensure_ascii=False) + '\n')
                except Exception as e:
                    print('ERROR:', e)
                time.sleep(0.3)
        print('Incremental done:', len(items), 'records')
    else:
        crawl_all()
