#!/usr/bin/env python3
"""crawl_jhlx_zjzwfw.py - 金华市兰溪市-建设项目环境影响评价信息公示 (浙江政务服务网)

站点: https://jhlx.zjzwfw.gov.cn/col/col1460389/index.html
CMS: 大汉版通 jcms1 经典版 (UTF-8, 服务端渲染, 无AJAX)
列表: a.bt_link[title=标题](href=/art/Y/M/D/art_1460389_ID.html) + div[style*=B4B4B4]日期
分页: 无分页, 固定9条(2025年停更)
详情: <title>/div.article_main 蓝底td 标题 + meta[pubDate] + div#zoom 正文
"""
import sys, os, requests, re, argparse
from urllib.parse import urljoin
from datetime import datetime, timedelta

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

CUTOFF = '2020-01-01'  # 停更站: 抓全部历史
SITE_NAME = '兰溪市-建设项目环评公示'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) Chrome/120.0'}
BASE = 'https://jhlx.zjzwfw.gov.cn'
LIST_URL = BASE + '/col/col1460389/index.html'

def fetch_list():
    try:
        r = requests.get(LIST_URL, timeout=(5, 15), headers=HEADERS, verify=False)
        r.encoding = 'utf-8'
    except:
        return []
    items = re.findall(
        r"<a href='([^']+)' class='bt_link' title='([^']*)'[^>]*>",
        r.text
    )
    results = []
    for href, title in items:
        title = re.sub(r'\s+', '', title).strip()
        if len(title) < 4:
            continue
        url = href if href.startswith('http') else BASE + href
        m = re.search(r"<a href='" + re.escape(href) + r"'.*?</a>.*?<div style=\"[^\"]*B4B4B4[^\"]*\">(\d{4}-\d{2}-\d{2})</div>", r.text, re.DOTALL)
        pd = m.group(1) if m else ''
        results.append({'url': url, 'pub_date': pd, 'title': title})
    return results

def extract_balanced_div(html, open_tag_re):
    """平衡 div 匹配"""
    m = re.search(open_tag_re, html)
    if not m:
        return ''
    gt = html.find('>', m.start())
    if gt == -1:
        return ''
    i = gt + 1
    depth = 1
    for mm in re.finditer(r'<div[\s>]|</div>', html[i:]):
        if mm.group(0).startswith('<div'):
            depth += 1
        else:
            depth -= 1
            if depth == 0:
                return html[i:mm.start() + i]
    return html[i:]

def fetch_detail(url):
    try:
        r = requests.get(url, timeout=(5, 15), headers=HEADERS, verify=False)
        r.encoding = 'utf-8'
    except:
        return None
    html = r.text
    title = ''
    m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
    if m:
        title = re.sub(r'\s+', '', m.group(1)).strip()
    if not title:
        m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
        if m:
            title = re.sub(r'\s*[-_—]\s*.*$', '', re.sub(r'\s+', '', m.group(1))).strip()
    pub_date = ''
    m = re.search(r'<meta\s+name="PubDate"\s+content="([^"]*)"', html)
    if m:
        pub_date = m.group(1)[:10]
    body = extract_balanced_div(html, r'<div[^>]+id="zoom"')
    if not body:
        body = extract_balanced_div(html, r'<div[^>]+id="Zoom"')
    if not body:
        body = extract_balanced_div(html, r'<td[^>]+id="bt_content"')
    body = re.sub(r'<script[^>]*>.*?</script>', '', body, flags=re.S | re.I)
    body = re.sub(r'<style[^>]*>.*?</style>', '', body, flags=re.S | re.I)
    def abs_link(m):
        href = m.group(1).strip()
        if href.startswith('http') or href.startswith('javascript') or href.startswith('#'):
            return m.group(0)
        if href.startswith('//'):
            return m.group(0).replace(href, 'https:' + href)
        return m.group(0).replace(href, BASE + href if href.startswith('/') else urljoin(url, href))
    import urllib.parse as urlparse
    body = re.sub(r'href="([^"]+)"', abs_link, body)
    body = re.sub(r'src="([^"]+)"', abs_link, body)
    body = re.sub(r'<p[^>]*>', '<p>', body)
    body = re.sub(r'<p>\s*</p>', '', body)
    text_len = len(re.sub(r'<[^>]+>', '', body).strip())
    has_img = '<img' in body
    if text_len < 5 and not has_img:
        return {'title': title, 'content': '', 'pub_date': pub_date}
    return {'title': title, 'content': body, 'pub_date': pub_date}

def main():
    max_pages = None
    sync_only = '--sync-only' in sys.argv
    for a in sys.argv[1:]:
        if a.startswith('--pages='):
            try:
                max_pages = int(a.split('=', 1)[1])
            except ValueError:
                pass
        elif a.isdigit():
            max_pages = int(a)
    if sync_only:
        return
    items = fetch_list()
    print(f'[jhlx] 列表 {len(items)} 条')
    results = []
    for it in items:
        if it['pub_date'] and it['pub_date'] < CUTOFF:
            continue
        detail = fetch_detail(it['url'])
        if not detail:
            continue
        if not detail['content']:
            detail = {'title': it['title'], 'publish_date': it['pub_date'], 'content': ''}
        detail['title'] = detail['title'] or it['title']
        detail['publish_date'] = detail.get('publish_date') or it['pub_date']
        results.append({
            'site_name': SITE_NAME,
            'title': detail['title'],
            'pub_date': detail['publish_date'],
            'content': detail['content'],
            'source_url': it['url'],
            'url': it['url'],
        })
        print(f"  {detail['publish_date']} {detail['title'][:40]}")
    push_to_searchdb(results, 'jhlx_zjzwfw')
    print(f'[jhlx] 入库 {len(results)} 条')

if __name__ == '__main__':
    main()
