#!/usr/bin/env python3
"""crawl_hhjs_jianshui.py - 建水县人民政府-环境质量状况 (西安博达VSB静态列表)"""

import sys, os, re, json, time, requests
from bs4 import BeautifulSoup

sys.stdout.reconfigure(line_buffering=True)
sys.stderr.reconfigure(line_buffering=True)

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

SITE_NAME = '建水县人民政府-环境质量状况'
DOMAIN = 'https://www.hhjs.gov.cn'
GROUP = '云南'
INDUSTRY = '生态环境'

# 列表页: 第1页 hjzlzk.htm, 第2-5页 /4.htm, /3.htm, /2.htm, /1.htm (倒序编号)
LIST_PATHS = [
    '/zfxxgk/fdzdgknr/zdlyxxgk/sthjgk/hjzlzk.htm',
    '/zfxxgk/fdzdgknr/zdlyxxgk/sthjgk/hjzlzk/4.htm',
    '/zfxxgk/fdzdgknr/zdlyxxgk/sthjgk/hjzlzk/3.htm',
    '/zfxxgk/fdzdgknr/zdlyxxgk/sthjgk/hjzlzk/2.htm',
    '/zfxxgk/fdzdgknr/zdlyxxgk/sthjgk/hjzlzk/1.htm',
]

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}


def fetch_list(session, list_path):
    """获取列表页，返回 [(href, title, pub_date), ...]"""
    url = DOMAIN + list_path
    try:
        resp = session.get(url, headers=HEADERS, timeout=20, verify=False)
        resp.encoding = 'utf-8'  # 防止Double-encoding
        if resp.status_code != 200:
            print(f'  × 列表页HTTP {resp.status_code}: {url}')
            return []
    except Exception as e:
        print(f'  × 列表页请求异常: {url} -> {e}')
        return []

    html = resp.text
    soup = BeautifulSoup(html, 'html.parser')
    items = []

    # 列表: <ul class="dot ulist">
    ulist = soup.find('ul', class_=lambda c: c and 'ulist' in c)
    if not ulist:
        print(f'  × 未找到ulist容器: {url}')
        return []

    for li in ulist.find_all('li'):
        a = li.find('a')
        if not a:
            continue
        href = a.get('href', '').strip()
        if not href or not href.endswith('.htm'):
            continue
        if not href.startswith('http'):
            href = DOMAIN + '/' + href.lstrip('./')

        # 标题: <div class="h4 eclip">
        title_div = a.find('div', class_=lambda c: c and 'h4' in c and 'eclip' in c)
        title = title_div.get_text(strip=True) if title_div else a.get_text(strip=True)

        # 日期: 从日期div或文本中提取
        date_str = ''
        for txt in li.stripped_strings:
            m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', txt)
            if m:
                date_str = m.group(1).replace('/', '-')
                break

        items.append((href, title, date_str))

    print(f'  √ 列表页: {len(items)} 条')
    return items


def fetch_detail(session, url):
    """获取详情页，返回 (title, pub_date, content_html)"""
    try:
        resp = session.get(url, headers=HEADERS, timeout=20, verify=False)
        resp.encoding = 'utf-8'  # 防止Double-encoding
        if resp.status_code != 200:
            return ('', '', '')
    except Exception as e:
        print(f'  × 详情页异常: {url} -> {e}')
        return ('', '', '')

    html = resp.text
    soup = BeautifulSoup(html, 'html.parser')

    # 标题: <div class="xxgk-tit"><h1>...</h1></div>
    title = ''
    tit_div = soup.find('div', class_=lambda c: c and 'xxgk-tit' in (c or ''))
    if tit_div:
        h1 = tit_div.find('h1')
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        t_tag = soup.find('title')
        if t_tag:
            title = t_tag.get_text(strip=True)
            title = re.sub(r'[-—].*?$', '', title).strip()

    # 发布日期: <meta name="PubDate" 或 页面日期
    pub_date = ''
    meta_date = soup.find('meta', attrs={'name': re.compile(r'PubDate|publishdate', re.I)})
    if meta_date and meta_date.get('content'):
        pub_date = meta_date['content'].strip()
    if not pub_date:
        for txt in soup.stripped_strings:
            m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', txt)
            if m:
                pub_date = m.group(1).replace('/', '-')
                break

    # 正文: <div class="xxgk-con"> -> <div id="vsb_content"> 或 TRS_Editor
    content_html = ''
    xxgk_con = soup.find('div', class_=lambda c: c and 'xxgk-con' in (c or ''))
    if xxgk_con:
        vsb = xxgk_con.find('div', id=re.compile(r'vsb_content'))
        if vsb:
            content_html = str(vsb)
        else:
            trs = xxgk_con.find('div', class_=lambda c: c and 'TRS_Editor' in (c or ''))
            if trs:
                content_html = str(trs)
            else:
                content_html = str(xxgk_con)

    if not content_html:
        content_html = str(soup.find('body') or '')

    # 提取保留 <p> <table> <a> <img>
    parts = []
    seen_keys = set()
    for tag_name in ['p', 'table', 'a', 'img']:
        for m in re.finditer(r'(?:<%s[^>]*/>|<%s[^>]*>.*?</%s>)' % (tag_name, tag_name, tag_name),
                             content_html, re.DOTALL):
            tag_html = m.group()
            key = re.sub(r'<[^>]+>', '', tag_html).strip()[:40]
            if key:
                if key not in seen_keys:
                    seen_keys.add(key)
                    parts.append(tag_html)
            elif '<img' in tag_html or '<a' in tag_html:
                parts.append(tag_html)

    if not parts:
        text = re.sub(r'<[^>]+>', '', content_html)
        text = re.sub(r'\s+', '\n', text).strip()
        return (title, pub_date, text)

    body_html = '\n'.join(parts)

    # 相对URL转绝对
    body_html = re.sub(
        r'(href|src)="(/(?!http)[^"]*)"',
        lambda m: '%s="%s%s"' % (m.group(1), DOMAIN, m.group(2)),
        body_html,
    )

    return (title, pub_date, body_html)


def run(incremental=False):
    session = requests.Session()
    # requests默认不验证SSL
    session.verify = False

    all_records = []
    seen = set()
    max_pages = 1 if incremental else len(LIST_PATHS)

    for page_idx in range(max_pages):
        items = fetch_list(session, LIST_PATHS[page_idx])
        if not items:
            continue
        for href, title, pub_date in items:
            if href not in seen:
                seen.add(href)
                all_records.append({
                    'title': title,
                    'url': href,
                    'pub_date': pub_date,
                    'site_name': SITE_NAME,
                })
        time.sleep(0.5)

    print(f'Total unique items: {len(all_records)}')

    # 获取详情
    for i, rec in enumerate(all_records):
        detail_title, detail_date, content = fetch_detail(session, rec['url'])
        rec['title'] = detail_title or rec['title']
        rec['pub_date'] = detail_date or rec['pub_date']
        if content:
            rec['content'] = content
            rec['summary'] = re.sub(r'<[^>]+>', '', content)[:200]
        else:
            rec['content'] = ''
            rec['summary'] = ''
        if (i + 1) % 10 == 0:
            print(f'  Detail {i+1}/{len(all_records)}')
        time.sleep(0.3)

    print(f'Detail fetched: {len(all_records)} items')
    if all_records:
        push_to_searchdb(all_records, 'hhjs_jianshui_hjzl')
    return len(all_records)


if __name__ == '__main__':
    inc = '--incremental' in sys.argv
    cnt = run(incremental=inc)
    print(f'Done: {cnt} records')
