#!/usr/bin/env python3
"""
扎鲁特旗人民政府 - 建设项目环境影响评价 爬虫
CMS: 内蒙古统一平台 (static pagination)
URL: http://www.zhalute.gov.cn/zwgk/zfxxgk/fdzdgknr/zdlyxx/sthj/jsxmhjyxpj/
"""

import sys, os, re, time, json
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import requests
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"

def get_connection():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

SITE_NAME = "扎鲁特旗-建设项目环评"
BASE_URL = "http://www.zhalute.gov.cn/zwgk/zfxxgk/fdzdgknr/zdlyxx/sthj/jsxmhjyxpj/"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

# Chinese date pattern: "2026年6月16日" or "2026年06月16日"
CN_DATE_RE = re.compile(r'(\d{4})年(\d{1,2})月(\d{1,2})日')
# Also try "2026-06-16" format
ISO_DATE_RE = re.compile(r'(\d{4}-\d{2}-\d{2})')


def extract_cn_date(text):
    """Extract date from Chinese date format like '2026年6月16日'"""
    m = CN_DATE_RE.search(text)
    if m:
        return "%s-%02d-%02d" % (m.group(1), int(m.group(2)), int(m.group(3)))
    m = ISO_DATE_RE.search(text)
    if m:
        return m.group(1)
    return ""


def crawl_list_page(page_num):
    """Crawl a single list page"""
    if page_num == 0:
        url = BASE_URL
    else:
        # index_1.html = page 1, index.html = page 0
        url = f"{BASE_URL}index_{page_num}.html"
    
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")
    except Exception as e:
        print(f"[List] Error page {page_num}: {e}")
        return []
    
    # Find all article links
    seen = set()
    items = []
    for a in soup.find_all("a"):
        href = a.get("href", "")
        txt = a.get_text(strip=True)
        if "zhalute" in href and "/t" in href and len(txt) > 10:
            if href not in seen:
                seen.add(href)
                date_str = extract_cn_date(txt)
                if not date_str:
                    date_str = ""
                items.append({
                    "title": txt,
                    "url": href,
                    "date_str": date_str,
                })
    return items


def crawl_detail(item):
    """Fetch detail page"""
    url = item["url"]
    try:
        resp = requests.get(url, headers=HEADERS, timeout=15)
        resp.encoding = "utf-8"
        soup = BeautifulSoup(resp.text, "html.parser")
    except Exception as e:
        print(f"[Detail] Error: {e}")
        return None
    
    # Title
    title = item["title"]
    mt = soup.find("meta", attrs={"name": "ArticleTitle"})
    if mt and mt.get("content"):
        title = mt["content"].strip()
    
    # Date
    date_str = item["date_str"]
    pd = soup.find("meta", attrs={"name": "PubDate"})
    if pd and pd.get("content"):
        date_str = pd["content"].strip()[:10]
    elif not date_str:
        date_str = extract_cn_date(title) or ""
    
    # Content - use lis_list or cont_cont
    body_html = ""
    for sel in ["div.lis_list", "div.cont_cont", "div#pare"]:
        div = soup.select_one(sel)
        if div:
            body_html = str(div)
            break
    
    if not body_html:
        # Try generic approach
        div = soup.find("div", class_=lambda c: c and ("cont" in str(c).lower() or "main" in str(c).lower() or "lis" in str(c).lower()) if c else False)
        if div:
            body_html = str(div)
    
    # Summary
    summary_text = ""
    if body_html:
        ts = BeautifulSoup(body_html, "html.parser")
        summary_text = ts.get_text(separator=" ", strip=True)[:300]
    
    return {
        "site_name": SITE_NAME,
        "title": title,
        "page_url": url,
        "publish_date": date_str,
        "source_url": url,
        "content": body_html,
        "summary": summary_text,
    }


def save_to_db(records):
    conn = get_connection()
    cursor = conn.cursor()
    inserted = 0
    for rec in records:
        try:
            cursor.execute("""
                INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, source_url, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)
            """, (
                rec["site_name"], rec["title"], rec["page_url"],
                rec["publish_date"], rec["source_url"], rec["content"],
                rec.get("summary", "")
            ))
            if cursor.rowcount > 0:
                inserted += 1
        except Exception as e:
            print(f"[DB] Error: {e}")
    conn.commit()
    conn.close()
    return inserted


def main():
    print(f"=== {SITE_NAME} ===")
    print(f"3年截止: {THREE_YEARS_AGO}")
    
    # Step 1: List pages 0-20 (within 3 years)
    print("\n--- Step 1: List pages ---")
    all_items = []
    for p in range(0, 21):  # index.html through index_20.html
        items = crawl_list_page(p)
        if not items:
            print(f"[List] Page {p}: empty")
            continue
        # Filter 3 years
        items = [it for it in items if it["date_str"] >= THREE_YEARS_AGO or not it["date_str"]]
        all_items.extend(items)
        dates = [it["date_str"] for it in items if it["date_str"]]
        date_range = f"{dates[0]}..{dates[-1]}" if dates else "?"
        print(f"[List] Page {p}: {len(items)} items ({date_range}) total: {len(all_items)}")
        time.sleep(0.3)
    
    print(f"\nTotal: {len(all_items)} items within 3 years")
    if not all_items:
        print("No items, exiting")
        return
    
    # Step 2: Details
    print(f"\n--- Step 2: {len(all_items)} details ---")
    records = []
    errors = 0
    
    with ThreadPoolExecutor(max_workers=3) as ex:
        fut_map = {ex.submit(crawl_detail, it): it for it in all_items}
        for i, fut in enumerate(as_completed(fut_map)):
            rec = fut.result()
            if rec:
                records.append(rec)
            else:
                errors += 1
            if (i + 1) % 20 == 0:
                print(f"[Detail] {i+1}/{len(all_items)} done")
    
    print(f"\nDetail: {len(records)} success, {errors} errors")
    
    # Step 3: Save
    print("\n--- Step 3: Save ---")
    inserted = save_to_db(records)
    print(f"Inserted {inserted} new")
    print("\n=== Done ===")


if __name__ == "__main__":
    main()
