#!/usr/bin/env python3
"""
crawl_gxhx.py - 横州市人民政府-普通通知公告
https://www.gxhx.gov.cn/yw/tzgg/pttzgg/
TRS WCM, 72页×20条/页
"""
import os, re, sys, time, urllib.request, urllib.error, urllib.parse
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "https://www.gxhx.gov.cn/yw/tzgg/pttzgg"
LIST_URL = BASE_URL + "/index.html"
HEADERS = {'User-Agent': 'Mozilla/5.0'}

CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
print(f"[gxhx] 3-year cutoff: {CUTOFF}")

# --- helpers ---
def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=20).read().decode('utf-8', errors='replace')
    except Exception as e:
        print(f"[gxhx] fetch fail {url}: {e}")
        return ""

def get_detail(url):
    """Extract title, date, body from detail page."""
    html = fetch(url)
    if not html:
        return None, None, None
    soup = BeautifulSoup(html, 'html.parser')
    
    # title
    title_el = soup.find('h2', id='contentTitle')
    title = title_el.get_text(strip=True) if title_el else ''
    
    # date
    date_el = soup.find('span', id='contentTime')
    date_str = ''
    if date_el:
        txt = date_el.get_text(strip=True)
        m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
        if m: date_str = m.group(1)
    
    # body - contentText div
    body_el = soup.find('div', id='contentText')
    body = ''
    if body_el:
        # 2026-08-13 修复: 剥掉页面style/script块 (动态类名导致内容每次重抓微变 + 防CSS泄漏)
        for st in body_el.find_all(['style', 'script']):
            st.decompose()
        # 2026-08-13 修复: 附件区不删除, 提取下载链接转为内嵌URL段落 (附件绝对化)
        attach_parts = []
        fj = body_el.find('div', class_='fj')
        if fj:
            for a in fj.find_all('a', href=True):
                abs_url = urllib.parse.urljoin(url, a.get('href', '').strip())
                name = a.get_text(strip=True) or os.path.basename(abs_url.split('?')[0]) or '附件'
                attach_parts.append('<p><a href="%s" target="_blank">%s</a></p>' % (abs_url, name))
            fj.decompose()
        apph = body_el.find('div', class_='apph')
        if apph: apph.decompose()
        # clear class/id on wrapper for clean inner content
        body = ''.join(str(c) for c in body_el.children).strip()
        if attach_parts:
            body = ((body + '\n\n') if body else '') + '\n\n'.join(attach_parts)
    
    return title, date_str, body.strip() if body else ''

# --- get total pages ---
html = fetch(LIST_URL)
page_match = re.search(r'createPageHTML\((\d+)', html)
if not page_match:
    print("[gxhx] Cannot find page count!")
    sys.exit(1)
total_pages = int(page_match.group(1))
print(f"[gxhx] Total pages: {total_pages}")

# --- parse all list pages ---
import sqlite3
conn = sqlite3.connect(DB_PATH, timeout=60)
cur = conn.cursor()

total_new = 0
total_skip = 0
total_external = 0

for page_idx in range(total_pages):
    if page_idx == 0:
        list_url = LIST_URL
    else:
        list_url = BASE_URL + f"/index_{page_idx}.html"
    
    html = fetch(list_url)
    if not html:
        print(f"[gxhx] Page {page_idx+1}: fetch fail, skip")
        continue
    
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='public-list')
    if not ul:
        print(f"[gxhx] Page {page_idx+1}: no public-list, skip")
        continue
    
    items = ul.find_all('li')
    print(f"[gxhx] Page {page_idx+1}/{total_pages}: {len(items)} items", end='')
    
    for li in items:
        a = li.find('a')
        if not a:
            continue
        
        href = a.get('href', '')
        title = a.get('title', a.get_text(strip=True))
        
        # date from list
        span = li.find('span', class_='time')
        date_str = ''
        if span:
            m = re.search(r'(\d{4}-\d{2}-\d{2})', span.get_text())
            if m: date_str = m.group(1)
        
        # filter by date
        if date_str and date_str < CUTOFF:
            continue
        
        # skip external links
        if href.startswith('http') and 'gxhx.gov.cn' not in href:
            total_external += 1
            continue
        
        # build full detail url
        if href.startswith('./'):
            detail_url = BASE_URL + '/' + href[2:]
        elif href.startswith('/'):
            detail_url = 'https://www.gxhx.gov.cn' + href
        elif not href.startswith('http'):
            detail_url = BASE_URL + '/' + href
        else:
            detail_url = href
        
        # get detail
        d_title, d_date, body = get_detail(detail_url)
        if not d_title:
            d_title = title
        
        if not d_date and date_str:
            d_date = date_str
        
        site_name = "横州市政府-普通通知公告"
        
        # check existing
        cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, site_name))
        if cur.fetchone():
            total_skip += 1
            continue
        
        body_clean = body if body else ''
        
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw (title, content, page_url, site_name, publish_date) VALUES (?,?,?,?,?)",
            (d_title, body_clean, detail_url, site_name, d_date)
        )
        if cur.rowcount > 0:
            total_new += 1
            if total_new % 20 == 0:
                conn.commit()
    
    print(f" → new:{total_new} skip:{total_skip} ext:{total_external}")
    conn.commit()
    time.sleep(0.3)

conn.commit()
conn.close()
print(f"[gxhx] DONE: {total_new} new, {total_skip} existing skipped, {total_external} external skipped")
