#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""2026-08-11 荆州 324 条 zw-leader-jl 半截残骸重抓修复:
旧版 clear_jz_noise 正则误删导致正文尾部截断 -> 重新抓取详情页恢复完整正文
"""
import re
import sqlite3
import time
import requests
from bs4 import BeautifulSoup

DB = '/mnt/data/search.db'
SITE = '荆州市生态环境局'
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Referer': 'http://jzssthjj.zwgk.jingzhou.gov.cn/',
}


def clean_title(t):
    t = re.sub(r'&middot;|&nbsp;|&#160;|\u200b|\ufeff', '', t or '')
    t = re.sub(r'^[•·]\s*', '', t)
    return re.sub(r'\s+', ' ', t).strip()


def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.raise_for_status()
        r.encoding = 'utf-8'
        soup = BeautifulSoup(r.text, 'html.parser')
    except Exception as e:
        print('  ERR', url, repr(e))
        return None
    # 标题
    title = ''
    # 优先: 页面 h1.zw-box-title（文章真实标题）
    h = soup.select_one('h1.zw-box-title')
    if h:
        title = clean_title(h.get_text())
    if not title:
        m = soup.find('meta', attrs={'name': 'ArticleTitle'})
        if m and m.get('content'):
            title = clean_title(m['content'])
    if not title:
        m = re.search(r'<title>(.*?)</title>', r.text, re.S)
        if m:
            # 格式: 真实标题-栏目-站名-政府信息公开
            t = re.sub(r'-(环境影响评价|荆州市生态环境局|政府信息公开)+$', '', m.group(1).strip())
            t = re.sub(r'^[^\-]*--?', '', t)
            title = clean_title(t)
    # 日期: 页面 zw-box-infos 发布时间 / meta PubDate / 任意 YYYY-MM-DD
    pub_date = ''
    infos = soup.select_one('div.zw-box-infos')
    if infos:
        m = re.search(r'(20\d{2}-\d{2}-\d{2})', infos.get_text())
        if m:
            pub_date = m.group(1)
    if not pub_date:
        m = soup.find('meta', attrs={'name': 'PubDate'})
        if m and m.get('content'):
            pub_date = m['content'].strip()[:10]
    # 正文容器: 荆州局模板用 <div class=zwnr> 或 TRS view
    body_el = None
    for sel in ['div.zw-content.zw-box', 'div.zw-content', 'div.zwnr', 'div.view.TRS_UEDITOR', 'div.view', 'div.TRS_UEDITOR']:
        body_el = soup.select_one(sel)
        if body_el:
            break
    if body_el is None:
        for el in soup.find_all(class_=re.compile('TRS_UEDITOR')):
            body_el = el
            break
    if body_el is None:
        print('  NO BODY', url)
        return {'title': title, 'content': '', 'pub_date': pub_date}

    # 附件: div.attachment 或 zw-leader-jl 相关附件
    atts = []
    # 页面级 attachment div
    page_att = soup.select_one('div.attachment')
    if page_att:
        for a in page_att.find_all('a', href=True):
            href = a['href'].strip()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|wps|7z)$', href, re.I) or 'download' in href.lower():
                name = clean_title(a.get_text(strip=True)) or href.split('/')[-1]
                abs_url = href if href.startswith('http') else ('https:' + href if href.startswith('//') else 'http://jzssthjj.zwgk.jingzhou.gov.cn' + (href if href.startswith('/') else '/' + href))
                atts.append({'name': name, 'url': abs_url})
    # zw-leader-jl 相关附件 div（含真实附件，保留）
    for jl in soup.find_all('div', class_='zw-leader-jl'):
        h3 = jl.find('h3')
        if h3 and '相关附件' in h3.get_text():
            for a in jl.find_all('a', href=True):
                href = a['href'].strip()
                if href:
                    name = clean_title(a.get_text(strip=True)) or href.split('/')[-1]
                    abs_url = href if href.startswith('http') else ('https:' + href if href.startswith('//') else 'http://jzssthjj.zwgk.jingzhou.gov.cn' + (href if href.startswith('/') else '/' + href))
                    if not any(x['url'] == abs_url for x in atts):
                        atts.append({'name': name, 'url': abs_url})

    # 正文: 从 body_el 提取，去掉空分类 zw-leader-jl（保留相关附件 div 转换的附件段落）
    body_html = str(body_el.decode_contents())
    # 删除 5 个空分类 zw-leader-jl div（相关链接/文档/图片/音频/视频）
    for jl in body_el.find_all('div', class_='zw-leader-jl'):
        h3 = jl.find('h3')
        if h3 and any(k in h3.get_text() for k in ['相关链接', '相关文档', '相关图片', '相关音频', '相关视频']):
            jl.decompose()
    # 相关附件 div: 转附件段落
    for jl in body_el.find_all('div', class_='zw-leader-jl'):
        h3 = jl.find('h3')
        if h3 and '相关附件' in h3.get_text():
            paras = []
            for a in jl.find_all('a', href=True):
                href = a['href'].strip()
                if href:
                    abs_url = href if href.startswith('http') else ('https:' + href if href.startswith('//') else 'http://jzssthjj.zwgk.jingzhou.gov.cn' + (href if href.startswith('/') else '/' + href))
                    name = clean_title(a.get_text(strip=True)) or href.split('/')[-1]
                    paras.append(f'<p><a href="{abs_url}">{name}</a></p>')
            if paras:
                jl.replace_with(BeautifulSoup('\n'.join(paras), 'html.parser'))
            else:
                jl.decompose()
    # 二维码/扫码/打印区
    for sel2 in ['div.jzgov-qrcode-div', 'div.jzgov-mt-40']:
        for el in body_el.select(sel2):
            el.decompose()
    for el in body_el.find_all(string=re.compile('扫一扫在手机上查看当前页面')):
        p = el.find_parent('div')
        if p:
            p.decompose()
    # 关闭按钮区
    for el in body_el.find_all(string=re.compile('关闭')):
        a = el.find_parent('a')
        if a and 'javascript' in (a.get('href') or ''):
            a.decompose()
    # 打印区
    for el in body_el.select('div.zw-box-print'):
        el.decompose()
    body_html = str(body_el.decode_contents())
    body_html = re.sub(r'<script[^>]*>.*?</script>', '', body_html, flags=re.S | re.I)
    body_html = re.sub(r'<style[^>]*>.*?</style>', '', body_html, flags=re.S | re.I)
    body_html = re.sub(r'<p[^>]*>', '<p>', body_html)
    # 图片/附件链接绝对化
    for a in body_el.find_all('a', href=True):
        href = a['href'].strip()
        if href.startswith('/') or href.startswith('./') or (not href.startswith('http') and not href.startswith('javascript')):
            a['href'] = ('https:' + href if href.startswith('//') else 'http://jzssthjj.zwgk.jingzhou.gov.cn' + (href if href.startswith('/') else '/' + href))
    for img in body_el.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith('http'):
            img['src'] = ('https:' + src if src.startswith('//') else 'http://jzssthjj.zwgk.jingzhou.gov.cn' + (src if src.startswith('/') else '/' + src))
    body_html = str(body_el.decode_contents())
    body_html = re.sub(r'<p[^>]*>', '<p>', body_html)
    body_html = re.sub(r'<p>\s*</p>', '', body_html)
    body_html = re.sub(r'\s*<div id=docAppendix.*$', '', body_html, flags=re.S)
    body_html = re.sub(r'\s*<div class=video.*$', '', body_html, flags=re.S)
    # 追加附件段落（若正文已含附件链接则不重复）
    has_att_in_body = any(a['url'] in body_html for a in atts)
    if atts and not has_att_in_body:
        att_html = ''.join(f'<p><a href="{a["url"]}">{a["name"]}</a></p>' for a in atts)
        body_html = body_html.rstrip() + '\n' + att_html

    text_len = len(re.sub(r'<[^>]+>', '', body_html).strip())
    has_img = '<img' in body_html
    if text_len < 5 and not has_img and not atts:
        return {'title': title, 'content': '', 'pub_date': pub_date}
    return {'title': title, 'content': body_html, 'pub_date': pub_date, 'attachments': atts}


def main():
    conn = sqlite3.connect(DB, timeout=60)
    conn.execute('PRAGMA busy_timeout=290000')
    rows = conn.execute(
        "SELECT id, page_url, title FROM gov_raw WHERE site_name=? AND content LIKE '%zw-leader-jl%'",
        (SITE,)).fetchall()
    print(f'待重抓: {len(rows)} 条')
    ok = fail = 0
    for rid, url, old_title in rows:
        det = fetch_detail(url)
        if det is None or not det.get('content'):
            print(f'  [{rid}] FAIL {old_title[:40]}')
            fail += 1
            continue
        try:
            conn.execute(
                'UPDATE gov_raw SET title=?, content=?, publish_date=? WHERE id=?',
                (det['title'] or old_title, det['content'], det['pub_date'] or '', rid))
            conn.commit()
            ok += 1
            print(f'  [{rid}] OK len={len(det["content"])} {old_title[:40]}')
        except Exception as e:
            print(f'  [{rid}] DBERR {e}')
            fail += 1
        time.sleep(0.5)
    print(f'完成: 成功 {ok} / 失败 {fail}')
    conn.close()


if __name__ == '__main__':
    main()
