#!/usr/bin/env python3
"""
crawl_hbdj.py - 杜集区人民政府-通知公告
TRS WCM/Ls.pagination分页，requests直取
"""
import os, re, sys, json, time, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import requests

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "杜集区人民政府-通知公告"
CATEGORY = "政府公告"
BASE = "https://www.hbdj.gov.cn"
LIST_API = BASE + "/content/column/14149471"
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
PAGE_SIZE = 20

print(f"[hbdj] 3-year cutoff: {CUTOFF_DATE}")

# Get page 1 to find total info
resp = requests.get(LIST_API + "?pageIndex=1", headers=HEADERS, timeout=30)
html = resp.text

# Extract total pages
page_count_match = re.search(r'pageCount[:\s]*(\d+)', html)
total_pages = int(page_count_match.group(1)) if page_count_match else 1
print(f"[hbdj] Total pages: {total_pages}")

conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()
new_count = dup_count = error_count = 0
skipped_old = 0

for pg in range(1, total_pages + 1):
    url = LIST_API + f"?pageIndex={pg}"
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        html = resp.text
    except Exception as e:
        print(f"  [WARN] Page {pg} error: {e}")
        error_count += PAGE_SIZE
        continue
    
    # Extract items via BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for li in soup.select('ul.doc_list li'):
        a = li.find('a', href=re.compile(r'xwzx/tzgg/\d+\.html'))
        span = li.find('span', class_='right date')
        if a and span is not None:
            href = a.get('href', '')
            title = a.get('title', '') or a.get_text(strip=True)
            date_str = span.get_text(strip=True)
            items.append((href, title, date_str))
    
    for href, title, date_str in items:
        if date_str < CUTOFF_DATE:
            skipped_old += 1
            continue
        
        detail_url = href if href.startswith("http") else BASE + href
        
        cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
        if cur.fetchone():
            dup_count += 1
            continue
        
        try:
            resp = requests.get(detail_url, headers=HEADERS, timeout=20)
            resp.encoding = 'utf-8'
            detail_html = resp.text
        except Exception as e:
            error_count += 1
            if error_count <= 3:
                print(f"  [WARN] Fetch error: {detail_url[:60]} -> {e}")
            continue
        
        d_title = title
        d_date = date_str
        body = ''
        
        mt = re.search(r'ArticleTitle"\s*content="([^"]+)"', detail_html)
        if mt and mt.group(1).strip():
            d_title = mt.group(1).strip()
        
        md = re.search(r'PubDate"\s*content="([^"]+)"', detail_html)
        if md:
            m = re.search(r'(\d{4}-\d{2}-\d{2})', md.group(1))
            if m:
                d_date = m.group(1)
        
        if d_date < CUTOFF_DATE:
            skipped_old += 1
            continue
        
        # Body in div#zoom.newscontnet — 2026-08-13 修复: 用BeautifulSoup提取完整内容
        # (原正则 (.*?)</div>\s*</div> 非贪婪, 正文含嵌套div/表格时在第一个</div>处截断, 丢失尾部段落)
        zsoup = BeautifulSoup(detail_html, 'html.parser')
        zdiv = zsoup.select_one('div#zoom.newscontnet') or zsoup.find('div', id='zoom')
        body = ''
        if zdiv:
            body = ''.join(str(c) for c in zdiv.contents).strip()

        # 2026-08-13 修复: 剥掉 lonsun_check_highlight 拼写检查 span (每字符一个 span, 污染正文)
        body = re.sub(r'<span[^>]*class="[^"]*lonsun_check_highlight[^"]*"[^>]*>(.*?)</span>', r'\1', body, flags=re.S | re.I)
        body = re.sub(r'<span[^>]*data-word-index[^>]*>(.*?)</span>', r'\1', body, flags=re.S | re.I)
        
        if not body:
            error_count += 1
            continue
        
        date_rank = int(d_date.replace("-", "")) if d_date else 0
        text_soup = BeautifulSoup(body, 'html.parser')
        summary = text_soup.get_text(strip=True)[:200]
        
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, d_title, detail_url, d_date, body, date_rank, CATEGORY, summary)
        )
        if cur.rowcount > 0:
            new_count += 1
    
    conn.commit()
    
    if pg % 10 == 0 or pg == total_pages:
        print(f"  Page {pg}/{total_pages}: +{new_count} new, {dup_count} dup, {error_count} err, {skipped_old} old")
    
    time.sleep(0.3)

conn.close()
print(f"\n[hbdj] Summary: +{new_count} new, {dup_count} dup, {error_count} err, {skipped_old} old")
print(json.dumps({"site": SITE_NAME, "new": new_count, "dup": dup_count, "err": error_count}, ensure_ascii=False))
