#!/usr/bin/env python3
"""crawl_nmgfzw.py - 内蒙古法治网-公告声明"""
import requests
import re
import sqlite3
import os
import sys
import json
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "内蒙古法治网-公告声明"
API_URL = "http://www.nmgfzw.org.cn/interface/LoadMores.php"
LIST_URL = "http://www.nmgfzw.org.cn/html/ggsm/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": "http://www.nmgfzw.org.cn/html/ggsm/"
}
CUTOFF_DATE = "2023-06-19"
MAX_PAGES = 30

session = requests.Session()
session.headers.update(HEADERS)

def fetch_page(page_num):
    """Fetch one page of items from API"""
    try:
        r = session.post(API_URL, data={
            "pagesize": page_num,
            "classid": 11,
            "pagecount": 10,
            "channel": "article"
        }, timeout=15)
        data = r.json()
        items = []
        for item in data:
            title = item.get("title", "").strip()
            url = item.get("arcurl", "")
            date_str = item.get("pubdate", "")
            if date_str < CUTOFF_DATE:
                continue
            items.append((url, title, date_str))
        return items
    except Exception as e:
        print(f"  Page {page_num} error: {e}", flush=True)
        return None

def extract_content(html):
    """Extract article content from detail page"""
    # Try div.content
    start = html.find('<div class="content"')
    if start < 0:
        start = html.find('class="content"')
        if start >= 0:
            prefix = html[:start]
            div_start = prefix.rfind('<div')
            if div_start >= 0:
                start = div_start
    
    if start < 0:
        return None
    
    # Find matching closing </div>
    depth = 0
    i = start
    in_tag = False
    tag_start = 0
    while i < len(html):
        if html[i] == '<':
            in_tag = True
            tag_start = i
        elif html[i] == '>':
            if in_tag:
                tag = html[tag_start:i+1]
                in_tag = False
                if tag.startswith('<!--'):
                    continue
                if tag.startswith('</div'):
                    depth -= 1
                    if depth == 0:
                        content = html[start:i+1]
                        # Clean up share buttons, scripts etc. at the end
                        # Remove content after </div> of share/share-box if present
                        return content.strip()
                elif not tag.startswith('<br') and not tag.startswith('<img') and not tag.startswith('<input') and not tag.startswith('<hr'):
                    if tag.startswith('<div'):
                        depth += 1
            in_tag = False
        i += 1
    return None

def fetch_detail(url):
    """Fetch detail page and extract content"""
    try:
        r = session.get(url, timeout=15)
        r.encoding = 'utf-8'
        return extract_content(r.text)
    except Exception as e:
        return None

def save_to_db(items):
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    
    new_count = 0
    skip_count = 0
    error_count = 0
    
    for idx, (url, title, date_str) in enumerate(items):
        try:
            content = fetch_detail(url)
            if not content:
                print(f"  [SKIP] No content: {title[:40]}", flush=True)
                error_count += 1
                continue
            
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, content) VALUES (?, ?, ?, ?, ?)",
                (SITE_NAME, title, url, date_str, content)
            )
            if c.rowcount > 0:
                new_count += 1
                if new_count % 20 == 0:
                    conn.commit()
            else:
                skip_count += 1
        except Exception as e:
            print(f"  [ERR] {title[:40]}: {e}", flush=True)
            error_count += 1
    
    conn.commit()
    conn.close()
    return new_count, skip_count, error_count

def main():
    print(f"=== {SITE_NAME} ===", flush=True)
    print(f"Cutoff: {CUTOFF_DATE}", flush=True)
    
    all_items = []
    
    # First: parse pre-rendered items from the first page
    try:
        r = session.get(LIST_URL, timeout=15)
        r.encoding = 'utf-8'
        for m in re.finditer(r'<li>(.*?)</li>', r.text, re.S):
            item_html = m.group(1)
            title_m = re.search(r'<h3[^>]*><a[^>]*href="([^"]+)"[^>]*>([^<]+)</a></h3>', item_html)
            date_m = re.search(r'(\d{4}-\d{2}-\d{2})', item_html)
            if title_m and date_m:
                url = title_m.group(1)
                title = title_m.group(2).strip()
                date_str = date_m.group(1)
                if date_str >= CUTOFF_DATE:
                    all_items.append((url, title, date_str))
        if all_items:
            print(f"  Pre-rendered: {len(all_items)} items (last: {all_items[-1][2]})", flush=True)
    except Exception as e:
        print(f"  First page error: {e}", flush=True)
    
    for page in range(1, MAX_PAGES + 1):
        items = fetch_page(page)
        if items is None:
            print(f"  Page {page}: error", flush=True)
            continue
        if not items:
            print(f"  Page {page}: empty (stopping)", flush=True)
            break
        all_items.extend(items)
        print(f"  Page {page}: {len(items)} items (last: {items[-1][2]})", flush=True)
        
        # Stop if all items are past cutoff
        if items and items[-1][2] < CUTOFF_DATE:
            break
    
    print(f"\nTotal items to process: {len(all_items)}", flush=True)
    
    if not all_items:
        print("No new items found.", flush=True)
        return
    
    new_count, skip_count, error_count = save_to_db(all_items)
    
    print(f"\n=== Summary ===", flush=True)
    print(f"New: {new_count}, Skipped: {skip_count}, Errors: {error_count}", flush=True)

if __name__ == "__main__":
    main()
