#!/usr/bin/env python3
"""
纳雍县人民政府 - 通知公告 (crawl_nayong_tzgg.py)
TRS CMS - 静态分页
总计3254条，163页
"""
import json, os, sys, time, re, urllib.request, sqlite3

SITE_NAME = "nayong_tzgg"
BASE_URL = "https://www.gznayong.gov.cn"
LIST_URL = f"{BASE_URL}/xwdt/tzgg"
TEMP_DB = "/root/temp_search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
TOTAL_PAGES = 163
PAGE_SIZE = 20


def fetch_list_page(page_num):
    """Fetch one list page"""
    if page_num == 1:
        url = f"{LIST_URL}/index.html"
    else:
        url = f"{LIST_URL}/index_{page_num}.html"
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    return resp.read().decode("utf-8", errors="replace")


def parse_list(html):
    """Extract articles from list page"""
    articles = []
    for m in re.finditer(
        r'<a[^>]*title="([^"]*)"[^>]*href="([^"]*)"[^>]*>.*?</a>\s*<span>([^<]*)</span>',
        html, re.DOTALL
    ):
        title = m.group(1).strip()
        href = m.group(2).strip()
        date = m.group(3).strip()
        if not href.startswith("http"):
            href = BASE_URL + href
        articles.append({"title": title, "url": href, "date": date})
    return articles


def fetch_detail(url):
    """Fetch article detail content"""
    try:
        req = urllib.request.Request(url, headers=HEADERS)
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode("utf-8", errors="replace")

        # Try Zoom first, then body_content
        content = ""
        m = re.search(r'<font[^>]*id="Zoom"[^>]*>(.*?)</font>', html, re.DOTALL)
        if m:
            content = m.group(1)
        else:
            m = re.search(r'<div[^>]*class="content"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
            if m:
                content = m.group(1)
            else:
                m = re.search(r'<div[^>]*class="body_content"[^>]*>(.*?)</div>', html, re.DOTALL)
                if m:
                    content = m.group(1)

        if content:
            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)

        text_len = len(re.sub(r'<[^>]+>', '', content).strip())
        if text_len < 30:
            return ""
        return content

    except urllib.error.HTTPError as e:
        if e.code == 404:
            return ""
        return ""
    except Exception:
        return ""


def main():
    os.chdir("/root")
    incremental = "--incremental" in sys.argv

    if incremental:
        pages_to_fetch = [1]
        print("Mode: INCREMENTAL (page 1)")
    else:
        pages_to_fetch = range(1, TOTAL_PAGES + 1)
        print(f"Mode: FULL ({TOTAL_PAGES} pages)")

    # Phase 1: Collect URLs
    print("=== Phase 1: Collecting article URLs ===")
    all_articles = []
    seen_urls = set()

    for page in pages_to_fetch:
        try:
            html = fetch_list_page(page)
            articles = parse_list(html)
            new = 0
            for art in articles:
                if art["url"] not in seen_urls:
                    seen_urls.add(art["url"])
                    all_articles.append(art)
                    new += 1
            print(f"  Page {page}/{TOTAL_PAGES}: {len(articles)} items, +{new} new, {len(all_articles)} total")
        except Exception as e:
            print(f"  Page {page} ERROR: {e}")
        time.sleep(0.2)

    total = len(all_articles)
    print(f"\n=== {total} unique articles collected ===")
    if total == 0:
        print("No articles found!")
        return

    # Save list
    with open("/tmp/nayong_articles.json", "w") as f:
        json.dump(all_articles, f, ensure_ascii=False)
    print("Saved to /tmp/nayong_articles.json")

    # Phase 2: Fetch details
    print("\n=== Phase 2: Fetching details ===")

    conn = sqlite3.connect(TEMP_DB)
    c = conn.cursor()
    c.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, content TEXT, source_url TEXT UNIQUE,
        site_name TEXT, pub_date TEXT
    )""")
    conn.commit()

    inserted = 0
    skipped = 0
    empty = 0
    last_save = -1

    for i, art in enumerate(all_articles):
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url = ?", (art["url"],))
        if c.fetchone()[0] > 0:
            skipped += 1
            if i % 100 == 0:
                print(f"  [{i}/{total}] done={inserted} skip={skipped} empty={empty}")
            continue

        content = fetch_detail(art["url"])
        if content:
            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, pub_date) VALUES (?, ?, ?, ?, ?)",
                    (art["title"], content, art["url"], SITE_NAME, art["date"])
                )
                conn.commit()
                inserted += 1
            except Exception as e:
                print(f"  DB error: {e}")
        else:
            empty += 1

        if i % 50 == 0 or i == total - 1:
            print(f"  [{i}/{total}] done={inserted} skip={skipped} empty={empty}")

        if i - last_save >= 100:
            with open("/tmp/nayong_progress.txt", "w") as f:
                f.write(str(i + 1))
            last_save = i

        time.sleep(0.15)

    conn.close()
    print(f"\n=== Done ===")
    print(f"Total: {total}")
    print(f"Inserted: {inserted}")
    print(f"Skipped: {skipped}")
    print(f"Empty content: {empty}")


if __name__ == "__main__":
    main()
