#!/usr/bin/env python3
"""Crawler for 江门市生态环境局 - 通知公告


URL: https://www.jiangmen.gov.cn/bmpd/jmssthjj/zwgk/tzgg/
Pattern: Static HTML pagination (index.html, index_2.html, ... index_20.html)
Content: <div id="zoomcon"> with full HTML (tables, paragraphs, attachments)
"""
import requests
import re
import sqlite3
import time
import random
import sys
import os

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
BASE_URL = "https://www.jiangmen.gov.cn/bmpd/jmssthjj/zwgk/tzgg"
SITE_NAME = "江门市生态环境局-通知公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
DELAY = (0.3, 0.5)
MAX_PAGES = 20  # total 20 pages, all within 3 years


def log(msg):
    print(f"[{time.strftime('%H:%M:%S')}] {msg}", flush=True)


def crawl_list(page):
    """Crawl a list page and extract article URLs + metadata."""
    if page == 1:
        url = f"{BASE_URL}/index.html"
    else:
        url = f"{BASE_URL}/index_{page}.html"
    
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        log(f"ERROR fetching list page {page}: {e}")
        return []
    
    html = r.text
    
    # Extract articles: <li><a href="..." title="标题">标题</a><span>日期</span></li>
    items = re.findall(
        r'<li><a href="([^"]*)"[^>]*title="([^"]*)"[^>]*>.*?</a><span>([^<]+)</span></li>',
        html,
        re.DOTALL,
    )
    
    results = []
    for href, title, date_str in items:
        # Make absolute URL
        if href.startswith("http"):
            full_url = href
        else:
            full_url = f"https://www.jiangmen.gov.cn{href}"
        
        results.append({
            "url": full_url,
            "title": title.strip(),
            "date": date_str.strip(),
        })
    
    return results


def crawl_detail(url):
    """Crawl a detail page and extract full content."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        log(f"ERROR fetching detail {url}: {e}")
        return None, None, None
    
    html = r.text
    
    # Extract meta
    title_m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    date_m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
    src_m = re.search(r'<meta name="ContentSource" content="([^"]*)"', html)
    
    title = title_m.group(1).strip() if title_m else ""
    pub_date = date_m.group(1).strip()[:10] if date_m else ""
    source = src_m.group(1).strip() if src_m else ""
    
    # Extract zoomcon content using depth counting
    m = re.search(r'id="zoomcon"', html)
    if not m:
        return title, pub_date, source, ""
    
    start = m.start()
    # Find the > after id="zoomcon"
    gt_pos = html.find(">", start)
    if gt_pos == -1:
        return title, pub_date, source, ""
    
    # Count depth starting from the opening <div>
    # Actually we're already inside the <div ...>, so start after >
    depth = 1
    i = gt_pos + 1
    while i < len(html) and depth > 0:
        if html[i:i+6] == "</div>":
            depth -= 1
            i += 6
        elif html[i:i+4] == "<div" and i + 4 < len(html) and html[i+4] in (' ', '>', '\n', '\t', '\r', "'", '"'):
            depth += 1
            i += 4
        else:
            i += 1
    
    content = html[start:i+6]  # include the opening tag through </div>
    return title, pub_date, source, content


def save_to_db(items):
    """Insert crawled items into production DB."""
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    cur = conn.cursor()
    
    inserted = 0
    for item in items:
        try:
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw 
                   (site_name, title, page_url, publish_date, source_url, content)
                   VALUES (?, ?, ?, ?, ?, ?)""",
                (
                    SITE_NAME,
                    item["title"],
                    item["url"],
                    item["date"],
                    item["source"],
                    item["content"],
                ),
            )
            if cur.rowcount > 0:
                inserted += 1
        except Exception as e:
            log(f"ERROR inserting {item['url'][:60]}: {e}")
    
    conn.commit()
    conn.close()
    return inserted


def main():
    log(f"Starting crawl for {SITE_NAME}")
    log(f"DB: {DB_PATH}")
    
    all_saved = []
    
    for page in range(1, min(MAX_PAGES, _MAX_PG or MAX_PAGES)+1):
        log(f"Page {page}/{MAX_PAGES}...")
        articles = crawl_list(page)
        
        if not articles:
            log(f"  No articles found on page {page}, stopping")
            break
        
        log(f"  Found {len(articles)} articles")
        page_items = []
        
        for i, art in enumerate(articles):
            time.sleep(random.uniform(*DELAY))
            result = crawl_detail(art["url"])
            if result is None:
                continue
            
            title, pub_date, source, content = result
            page_items.append({
                "url": art["url"],
                "title": title or art["title"],
                "date": pub_date or art["date"],
                "source": source or SITE_NAME.replace("-通知公告", ""),
                "content": content or "",
            })
            
            if (i + 1) % 10 == 0:
                log(f"  Detail {i+1}/{len(articles)}")
        
        # Save this page's items
        if page_items:
            saved = save_to_db(page_items)
            all_saved.append(len(page_items))
            log(f"  Saved {saved}/{len(page_items)} new items (page {page})")
    
    total = sum(all_saved)
    log(f"\n=== SUMMARY ===")
    log(f"Total pages crawled: {len(all_saved)}")
    log(f"Total items saved: {total}")
    log(f"Done!")


if __name__ == "__main__":
    main()
