#!/usr/bin/env python3
"""crawl_htz_bengbu.py - 蚌埠高新区-项目环境评价"""
import re
import json
import os
import sys
import sqlite3
from datetime import datetime, timedelta
from playwright.sync_api import sync_playwright

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "蚌埠高新区-项目环境评价"
LIST_URL = "https://htz.bengbu.gov.cn/zfxxgk/public/column/29651?type=4&catId=7881031&action=list"
CUTOFF_DATE = "2023-06-19"
MAX_PAGES = 100  # safety limit

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def extract_items_from_html(html):
    """Extract article links, titles, dates from rendered list HTML"""
    items = []
    for m in re.finditer(
        r'<a href="(https://htz\.bengbu\.gov\.cn/zfxxgk/public/29651/\d+\.html)"[^>]*title="([^"]*)"[^>]*>.*?</a>\s*<span class="date">(\d{4}-\d{2}-\d{2})',
        html, re.S
    ):
        url = m.group(1)
        title = m.group(2).strip()
        date_str = m.group(3)
        if date_str < CUTOFF_DATE:
            continue
        items.append((url, title, date_str))
    return items

def extract_content(html):
    """Extract content from detail page"""
    # Try j-fontContent gkwz_contnet
    start = html.find('<div class="j-fontContent gkwz_contnet"')
    if start < 0:
        # Try xxgkcontent
        start = html.find('<div class="clearfix xxgkcontent')
        if start < 0:
            # Try gkwz_content
            start = html.find('class="clearfix gkwz_content')
            if start < 0:
                return None
            # Find opening <div
            prefix = html[:start]
            div_start = prefix.rfind('<div')
            if div_start >= 0:
                start = div_start
    
    if start < 0:
        return None
    
    # Find matching closing </div> with nesting
    depth = 0
    i = start
    in_tag = False
    tag_start = 0
    while i < len(html):
        if html[i] == '<':
            in_tag = True
            tag_start = i
        elif html[i] == '>':
            if in_tag:
                tag = html[tag_start:i+1]
                in_tag = False
                if tag.startswith('<!--'):
                    continue
                if tag.startswith('</div'):
                    depth -= 1
                    if depth == 0:
                        content = html[start:i+1]
                        # Clean up
                        content = re.sub(r'^\s*(?:<p[^>]*>\s*(?:&nbsp;|\s)*\s*</p>\s*|<br\s*/?>\s*)*', '', content)
                        return content.strip()
                elif not tag.startswith('<br') and not tag.startswith('<img') and not tag.startswith('<input') and not tag.startswith('<hr'):
                    if tag.startswith('<div') or tag.startswith('<DIV'):
                        depth += 1
            in_tag = False
        elif html[i] == '\n':
            if in_tag:
                # Check for comment
                pass
        i += 1
    return None

def main():
    print(f"=== {SITE_NAME} ===", flush=True)
    print(f"Cutoff: {CUTOFF_DATE}", flush=True)
    
    all_items = []
    
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--no-sandbox"])
        page = browser.new_page()
        
        # Load list page
        print("Loading list page...", flush=True)
        page.goto(LIST_URL, timeout=30000)
        page.wait_for_timeout(3000)
        
        # First page
        html = page.content()
        items = extract_items_from_html(html)
        if items:
            all_items.extend(items)
            print(f"  Page 1: {len(items)} items (first: {items[0][2]}, last: {items[-1][2]})", flush=True)
        
        # Get total count from body
        body_text = page.inner_text("body")
        total_match = re.search(r'共\s*(\d+)\s*条', body_text)
        total_items = int(total_match.group(1)) if total_match else 100
        print(f"  Total items: {total_items}", flush=True)
        
        # Click through remaining pages
        for page_num in range(2, MAX_PAGES + 1):
            # Find and click page number link
            clicked = False
            links = page.query_selector_all("a")
            for link in links:
                text = link.inner_text().strip()
                if text == str(page_num):
                    link.click()
                    page.wait_for_timeout(2000)
                    html = page.content()
                    items = extract_items_from_html(html)
                    if items:
                        all_items.extend(items)
                        print(f"  Page {page_num}: {len(items)} items (last date: {items[-1][2]})", flush=True)
                    clicked = True
                    break
            
            if not clicked:
                # Try "下一页"
                for link in links:
                    text = link.inner_text().strip()
                    if text == "下一页":
                        link.click()
                        page.wait_for_timeout(2000)
                        html = page.content()
                        items = extract_items_from_html(html)
                        if items:
                            all_items.extend(items)
                            print(f"  Page {page_num} (next): {len(items)} items (last date: {items[-1][2]})", flush=True)
                        clicked = True
                        break
            
            if not clicked:
                print(f"  No more pages at page {page_num}", flush=True)
                break
            
            # Stop if all items are before cutoff
            if items and items[-1][2] < CUTOFF_DATE:
                print(f"  Reached cutoff date, stopping", flush=True)
                break
        try:
            browser.close()
        except:
            pass
            pass
    
    print(f"\nTotal items collected: {len(all_items)}", flush=True)
    
    if not all_items:
        print("No new items found.", flush=True)
        return
    
    # Now fetch detail content
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    new_count = 0
    skip_count = 0
    error_count = 0
    error_titles = []
    
    with sync_playwright() as p:
        browser = p.chromium.launch(headless=True, args=["--no-sandbox"])
        page = browser.new_page()
        
        for idx, (url, title, date_str) in enumerate(all_items):
            try:
                page.goto(url, timeout=30000)
                page.wait_for_timeout(1500)
                html = page.content()
                
                content = extract_content(html)
                if not content:
                    print(f"  [SKIP] No content: {title[:40]}", flush=True)
                    error_count += 1
                    error_titles.append(title[:50])
                    continue
                
                # Also extract title from detail page
                title_m = re.search(r'<title>(.*?)</title>', html, re.S)
                detail_title = title_m.group(1).strip() if title_m else title
                # Remove suffix
                detail_title = re.sub(r'_蚌埠高新技术产业开发区管理委员会$', '', detail_title).strip()
                
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, content) VALUES (?, ?, ?, ?, ?)",
                    (SITE_NAME, detail_title, url, date_str, content)
                )
                if c.rowcount > 0:
                    new_count += 1
                    if new_count % 20 == 0:
                        conn.commit()
                        print(f"  Progress: {new_count}/{len(all_items)}", flush=True)
                else:
                    skip_count += 1
                    
            except Exception as e:
                print(f"  [ERR] {title[:40]}: {e}", flush=True)
                error_count += 1
                error_titles.append(title[:50])
        
            try:
                browser.close()
            except:
                pass
    
    conn.commit()
    conn.close()
    
    print(f"\n=== Summary ===", flush=True)
    print(f"New: {new_count}, Skipped: {skip_count}, Errors: {error_count}", flush=True)
    if error_titles:
        print(f"Error items (first 5):", flush=True)
        for t in error_titles[:5]:
            print(f"  - {t}", flush=True)

if __name__ == "__main__":
    main()
