#!/usr/bin/env python3
"""
内蒙古自治区发改委 - 通知公告 爬虫
https://fgw.nmg.gov.cn/xxgk/zxzx/tzgg/
TRS WCM, JS pagination, 34 pages x 15 items
"""

import requests
import re
import sqlite3
import os
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

DB_PATH = "/root/search.db"
BASE_URL = "https://fgw.nmg.gov.cn"
LIST_URL = "https://fgw.nmg.gov.cn/xxgk/zxzx/tzgg/"
SITE_NAME = "内蒙古自治区发改委-通知公告"
CUTOFF_DATE = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
CATEGORY = "通知公告"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

def get_total_pages():
    """Extract countPage from JS inside s_fenye div"""
    resp = requests.get(LIST_URL + "index.html", headers=HEADERS, timeout=30)
    resp.encoding = 'utf-8'
    
    m = re.search(r'var countPage\s*=\s*(\d+)', resp.text)
    if m:
        total = int(m.group(1))
        print(f"Found countPage = {total} from JS")
        return total
    
    return 50  # fallback

def parse_list_page(page_num):
    """
    Parse list page. 0-indexed:
    page_num=0 -> index.html
    page_num>=1 -> index_{page_num}.html
    Returns [(title, full_url, date_str)]
    """
    if page_num == 0:
        url = LIST_URL + "index.html"
    else:
        url = f"{LIST_URL}index_{page_num}.html"
    
    resp = requests.get(url, headers=HEADERS, timeout=30)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    items = []
    ul = soup.find('ul', class_='newsList')
    if not ul:
        return items
    
    for li in ul.find_all('li'):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href']
        title = a.get_text(strip=True)
        if len(title) < 5:
            continue
        
        # Build absolute URL from relative path
        if href.startswith('./'):
            full_url = f"{LIST_URL}{href[2:]}"
        elif href.startswith('/'):
            full_url = f"{BASE_URL}{href}"
        elif href.startswith('http'):
            full_url = href
        else:
            full_url = f"{LIST_URL}{href}"
        
        # Get date from span
        date_str = ""
        span = li.find('span')
        if span:
            dm = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', span.get_text())
            if dm:
                date_str = dm.group(1).replace('/', '-')
        
        items.append((title, full_url, date_str))
    
    return items

def fetch_detail(url):
    """Get detail: (full_title, content_html, date)"""
    resp = requests.get(url, headers=HEADERS, timeout=30)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    # Full title
    title_tag = soup.find('title')
    full_title = title_tag.get_text(strip=True) if title_tag else "无标题"
    full_title = re.sub(r'[-_\s]*内蒙古自治区发展和改革委员会.*', '', full_title).strip()
    if not full_title:
        full_title = "无标题"
    
    # Content
    content_div = soup.find('div', class_='TRS_Editor')
    if not content_div:
        content_div = soup.find('div', class_=re.compile(r'(?i)content|article|text'))
    if not content_div:
        content_div = soup.find('div', id=re.compile(r'(?i)zoom|content|article'))
    
    content_html = ""
    if content_div:
        for tag in content_div.find_all(['script', 'style', 'iframe']):
            tag.decompose()
        # Remove share button noise
        for share in content_div.find_all('div', class_='article-share-group'):
            share.decompose()
        for share in content_div.find_all('div', class_='share-box'):
            share.decompose()
        for info in content_div.find_all('div', class_='xl-info'):
            info.decompose()
        for h1 in content_div.find_all('h1', class_='xl-title'):
            h1.decompose()
        for pv in content_div.find_all('span', id='pageview'):
            pv.decompose()
        content_html = str(content_div)
    
    # Date from meta
    date_str = ""
    for meta in soup.find_all('meta'):
        name = meta.get('name', '').lower()
        if name in ('publishdate', 'pubdate', 'dc.date', 'date'):
            date_str = meta.get('content', '')
            break
    if not date_str:
        for meta in soup.find_all('meta'):
            if meta.get('itemprop') == 'datePublished':
                date_str = meta.get('content', '')
                break
    if not date_str:
        dm = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', resp.text)
        if dm:
            date_str = dm.group(1).replace('/', '-')
    
    if date_str:
        date_str = date_str.replace('/', '-')[:10]
    
    return full_title, content_html, date_str

def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    
    total_pages = get_total_pages()
    print(f"Total pages: {total_pages}")
    
    existing = set()
    for row in c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        existing.add(row[0])
    print(f"Existing items in DB: {len(existing)}")
    
    all_items = []
    seen_urls = set()
    
    for p in range(total_pages):  # 0-indexed: 0..total_pages-1
        items = parse_list_page(p)
        if not items:
            print(f"  Page {p+1}: 0 items (stopping)")
            break
        
        new_count = 0
        for title, url, date_str in items:
            if url in seen_urls or url in existing:
                continue
            seen_urls.add(url)
            if date_str and date_str < CUTOFF_DATE:
                continue
            all_items.append((title, url, date_str))
            new_count += 1
        
        print(f"  Page {p+1}/{total_pages}: {len(items)} items, {new_count} new in range")
    
    print(f"\nTotal items for detail: {len(all_items)}")
    
    inserted = 0
    for i, (title, page_url, list_date) in enumerate(all_items):
        try:
            full_title, content_html, detail_date = fetch_detail(page_url)
            final_date = detail_date or list_date
            final_title = full_title if full_title and len(full_title) >= len(title) else title
            
            date_rank = 0
            if final_date:
                try:
                    dt = datetime.strptime(final_date, "%Y-%m-%d")
                    date_rank = int(dt.timestamp())
                except:
                    pass
            
            c.execute("""INSERT OR IGNORE INTO gov_raw 
                        (site_name, title, page_url, publish_date, content, date_rank, category, summary)
                        VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
                     (SITE_NAME, final_title, page_url, final_date, content_html, date_rank, CATEGORY, ""))
            
            if c.rowcount > 0:
                inserted += 1
            
            if (i + 1) % 50 == 0:
                conn.commit()
                print(f"  Progress: {i+1}/{len(all_items)}, inserted: {inserted}")
        except Exception as e:
            print(f"  Error {page_url}: {e}")
    
    conn.commit()
    print(f"\nDone. Inserted {inserted} new items for {SITE_NAME}")
    conn.close()

if __name__ == "__main__":
    main()
