#!/usr/bin/env python3
"""
crawl_henanlt.py — 河南蓝天环境工程有限公司-公示公告
Site: https://henanlt.com/news/4
CMS: Custom PHP (ThinkPHP)
List: ul.news > li > a[title][href]
Detail: div.content (rich HTML + file attachments)
Pagination: /news/news_type/type/4/p/{page}
"""
import re, os, sys, time, sqlite3, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "henanlt.com-公示公告"
BASE_URL = "https://henanlt.com"
LIST_URL = BASE_URL + "/news/4"
PAGE_URL_TPL = BASE_URL + "/news/news_type/type/4/p/{}"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": BASE_URL + "/",
}

session = requests.Session()
session.headers.update(HEADERS)


def fetch(url):
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERR] fetch failed: {url} - {e}")
        return None


def parse_list(html):
    """Parse list page, return list of (url, title)"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='news')
    if not ul:
        return items
    for li in ul.find_all('li'):
        a = li.find('a')
        if a and a.get('href'):
            href = a['href']
            title = a.get('title') or a.get_text(strip=True)
            if not href.startswith('http'):
                href = BASE_URL + href
            items.append((href, title.strip()))
    return items


def parse_detail(html, url):
    """Parse detail page, return (title, publish_date, content)"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from h2 inside #gsjjnr
    gsjjnr = soup.find('div', id='gsjjnr')
    if not gsjjnr:
        return None, None, None
    
    # Title
    h2 = gsjjnr.find('h2')
    title = h2.get_text(strip=True) if h2 else None
    
    # Date from new_fb div: "发布时间：2026-05-29 15:45:06"
    new_fb = gsjjnr.find('div', class_='new_fb')
    date_str = None
    if new_fb:
        m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', new_fb.get_text())
        if m:
            date_str = m.group(1)
    
    # Content from div.content
    content_div = gsjjnr.find('div', class_='content')
    content = ''
    if content_div:
        content = str(content_div)
        # Clean up
        content = content.strip()
    
    return title, date_str, content


def get_total_pages():
    """Get total pages from page 1"""
    html = fetch(LIST_URL)
    if not html:
        return 0
    
    # "301 条记录 1/34 页"
    m = re.search(r'(\d+)\s*条记录\s*\d+/(\d+)\s*页', html)
    if m:
        total_items = int(m.group(1))
        total_pages = int(m.group(2))
        print(f"  Found: {total_items} items, {total_pages} pages")
        return total_pages
    
    # Fallback: check select options
    m = re.search(r'<option[^>]*value="(\d+)"[^>]*>', html)
    if m:
        # Last option value
        options = re.findall(r'<option[^>]*value="(\d+)"', html)
        if options:
            return int(options[-1])
    
    return 1


def main():
    print(f"=== {SITE_NAME} ===")
    print(f"Cutoff: {CUTOFF}")
    
    total_pages = get_total_pages()
    if total_pages == 0:
        print("ERROR: Could not determine total pages")
        return
    
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    cur = conn.cursor()
    
    total_new = 0
    total_skip = 0
    
    for page in range(1, total_pages + 1):
        if page == 1:
            list_url = LIST_URL
        else:
            list_url = PAGE_URL_TPL.format(page)
        
        print(f"\nPage {page}/{total_pages}: {list_url}")
        html = fetch(list_url)
        if not html:
            print(f"  [SKIP] page {page} fetch failed")
            continue
        
        items = parse_list(html)
        if not items:
            print(f"  [SKIP] page {page} no items")
            continue
        
        page_new = 0
        for url, title in items:
            # Check if already in DB
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                total_skip += 1
                continue
            
            # Fetch detail
            detail_html = fetch(url)
            if not detail_html:
                total_skip += 1
                continue
            
            full_title, date_str, content = parse_detail(detail_html, url)
            if not full_title and not title:
                total_skip += 1
                continue
            
            if not full_title:
                full_title = title
            
            if not date_str:
                total_skip += 1
                continue
            
            # Filter by date
            if date_str < CUTOFF:
                total_skip += 1
                continue
            
            # Content validation
            content = content.strip()
            if not content:
                total_skip += 1
                continue
            
            # Summary (plaintext first ~200 chars)
            summary = ''
            if content:
                text_soup = BeautifulSoup(content, 'html.parser')
                plain = text_soup.get_text(strip=True)
                summary = plain[:200]
            
            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, title, page_url, publish_date, content, date_rank, category) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, full_title, url, date_str, content,
                     int(date_str.replace("-", "")), 'tzgg')
                )
                if cur.rowcount > 0:
                    page_new += 1
                    total_new += 1
            except Exception as e:
                print(f"  [ERR] insert: {e}")
        
        conn.commit()
        print(f"  Page {page}: +{page_new} new (total {total_new}, skip {total_skip})")
        
        # Be polite
        time.sleep(0.5)
    
    conn.close()
    print(f"\n=== Done: {total_new} new, {total_skip} skipped ===")


if __name__ == "__main__":
    main()
