#!/usr/bin/env python3
"""
crawl_hhpb.py - 屏边苗族自治县人民政府-公示公告
https://www.hhpb.gov.cn/zfxx/gsgg.htm
CMS: VSB 9 (Visual SiteBuilder 9)
Detail: div#vsb_content, meta ArticleTitle/PubDate, list ul.dot.ulist
Pagination: gsgg/{N}.htm (reverse numbering: page N -> gsgg/{6-N}.htm)
"""
import os, re, sys, json, time, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE = "https://www.hhpb.gov.cn"
LIST_BASE = BASE + "/zfxx/gsgg"
LIST_PAGE1 = LIST_BASE + ".htm"
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
SITE_NAME = "屏边县政府-公示公告"
CATEGORY = "环保公示"

HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}

print(f"[hhpb] 3-year cutoff: {CUTOFF_DATE}")

def fetch(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"[hhpb] fetch fail {url}: {e}")
        return ""

def parse_list(html):
    """Parse ul.dot.ulist from list page"""
    items = []
    m = re.search(r'<ul class="dot ulist">(.*?)</ul>', html, re.DOTALL)
    if not m:
        return items
    ul_html = m.group(1)
    lis = re.findall(r'<li>(.*?)</li>', ul_html, re.DOTALL)
    for li in lis:
        href_m = re.search(r'href="([^"]*)"', li)
        title_m = re.search(r'<div class="h4 eclip">(.*?)</div>', li)
        date_m = re.search(r'<div class="time">(.*?)</div>', li)
        if href_m and title_m and date_m:
            href = href_m.group(1)
            title = BeautifulSoup(title_m.group(1), 'html.parser').get_text(strip=True)
            date_str = date_m.group(1).strip()
            items.append((title, href, date_str))
    return items

def get_detail(url):
    """Extract title, date, body from VSB 9 detail page"""
    html = fetch(url)
    if not html:
        return None, None, None
    
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from meta
    title = ''
    mt = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if mt and mt.get('content'):
        title = mt['content'].strip()
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)
    
    # Date from meta
    date_str = ''
    md = soup.find('meta', attrs={'name': 'PubDate'})
    if md and md.get('content'):
        m = re.search(r'(\d{4}-\d{2}-\d{2})', md['content'])
        if m:
            date_str = m.group(1)
    
    # Content from div#vsb_content inside div.xxgk-con
    body_parts = []
    
    vsb = soup.select_one('div[id^="vsb_content"]')
    if vsb:
        body_parts.append(str(vsb))
    
    # Check for attachment section after the content
    # Find the xxgk-con that contains vsb_content and check for attachments
    xxgk_cons = soup.find_all('div', class_='xxgk-con')
    # The last xxgk-con might have attachments
    if xxgk_cons:
        # vsb_content is in the first xxgk-con
        # Check if there are more xxgk-con with attachments
        pass
    
    body = '\n'.join(body_parts) if body_parts else ''
    return title, date_str, body

# ── Get total pages from page 1 ──
html1 = fetch(LIST_PAGE1)
items = parse_list(html1)
print(f"[hhpb] Page 1: {len(items)} items")

# Find total pages from pagination
total_pages = 5  # default from observed data
pag_m = re.search(r'class="p_last p_fun"[^>]*><a href="gsgg/(\d+)\.htm">尾页</a>', html1)
if pag_m:
    last_page_num = int(pag_m.group(1))
    total_pages = 5  # we know it's 5 pages from observation
    print(f"[hhpb] Total pages: {total_pages}")
else:
    print(f"[hhpb] Defaulting to 1 page")

# ── Collect all items from all pages ──
all_items = items  # start with page 1 items
for page_idx in range(2, total_pages + 1):
    # Reverse numbering: page 2 -> gsgg/4.htm, page 3 -> gsgg/3.htm, etc.
    url_num = total_pages + 1 - page_idx  # page 2 -> 4, page 3 -> 3, etc.
    page_url = f"{LIST_BASE}/{url_num}.htm"
    h = fetch(page_url)
    if h:
        page_items = parse_list(h)
        all_items.extend(page_items)
        print(f"[hhpb] Page {page_idx} ({url_num}.htm): {len(page_items)} items")
    time.sleep(0.5)

print(f"[hhpb] Total items from all pages: {len(all_items)}")

# ── Filter by cutoff date ──
active = [(t, u, d) for t, u, d in all_items if d >= CUTOFF_DATE]
skipped = len(all_items) - len(active)
print(f"[hhpb] Within 3yr: {len(active)}, older: {skipped}")

# ── Fetch details ──
import sqlite3
conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()

new_count = dup_count = error_count = 0

for idx, (title, href, date_str) in enumerate(active, 1):
    # Build absolute URL
    # href is like "../info/8621/541741.htm" or "info/8621/541741.htm"
    if href.startswith("http"):
        detail_url = href
    elif href.startswith("../"):
        detail_url = BASE + href[2:]  # ../info/... -> /info/...
    elif href.startswith("/"):
        detail_url = BASE + href
    else:
        detail_url = BASE + "/" + href
    
    try:
        cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
        if cur.fetchone():
            dup_count += 1
            continue
        
        d_title, d_date, body = get_detail(detail_url)
        if not d_title:
            d_title = title
        if not d_date:
            d_date = date_str
        
        date_rank = int(d_date.replace("-", "")) if d_date and "-" in d_date else 0
        summary = ""
        if body:
            text_soup = BeautifulSoup(body, 'html.parser')
            plain = text_soup.get_text(strip=True)
            summary = plain[:200]
        
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (SITE_NAME, d_title, detail_url, d_date, body, date_rank, CATEGORY, summary)
        )
        if cur.rowcount > 0:
            new_count += 1
    except Exception as e:
        error_count += 1
        print(f"  ERROR [{idx}] {title[:40]}: {e}")
    
    conn.commit()
    time.sleep(0.3)
    
    if idx % 10 == 0 or idx == len(active):
        print(f"  Progress {idx}/{len(active)}: +{new_count} new, {dup_count} dup, {error_count} err")

conn.close()
print(f"\n[hhpb] Summary: +{new_count} new, {dup_count} dup, {error_count} err")
print(json.dumps({"site": SITE_NAME, "new": new_count, "dup": dup_count, "err": error_count, "total": len(active)}, ensure_ascii=False))
