#!/usr/bin/env python3
"""
正安县人民政府 - 公示公告 (crawl_gzza_tzgg.py)
TRS CMS - 静态分页
总计~2790条(186页, 15条/页)
"""
import json, os, sys, time, re, urllib.request, sqlite3

SITE_NAME = "gzza_tzgg"
BASE_URL = "https://www.gzza.gov.cn"
LIST_URL = f"{BASE_URL}/xwzx/tzgg"
TEMP_DB = "/root/search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
TOTAL_PAGES = 186


def fetch_list_page(page_num):
    if page_num == 1:
        url = f"{LIST_URL}/index.html"
    else:
        url = f"{LIST_URL}/index_{page_num}.html"
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    return resp.read().decode("utf-8", errors="replace")


def parse_list(html):
    articles = []
    # TRS pattern: <a TARGET="_blank" title="TITLE" href="URL">TITLE</a><i>DATE</i>
    for m in re.finditer(
        r'<a[^>]*title="([^"]*)"[^>]*href="([^"]*)"[^>]*>.*?<i>([^<]*)</i>',
        html, re.DOTALL
    ):
        title = m.group(1).strip()
        href = m.group(2).strip()
        date = m.group(3).strip()[:10]
        if not href.startswith("http"):
            href = BASE_URL + href
        articles.append({"title": title, "url": href, "date": date})
    return articles


def fetch_detail(url):
    try:
        req = urllib.request.Request(url, headers=HEADERS)
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode("utf-8", errors="replace")
        content = ""
        m = re.search(r'<font[^>]*id="Zoom"[^>]*>(.*?)</font>', html, re.DOTALL)
        if m:
            content = m.group(1)
        else:
            m = re.search(r'<div[^>]*class="zx_content"[^>]*>(.*?)</div>', html, re.DOTALL)
            if m:
                content = m.group(1)
        if content:
            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
        text_len = len(re.sub(r'<[^>]+>', '', content).strip())
        return content if text_len >= 30 else ""
    except urllib.error.HTTPError as e:
        return "" if e.code == 404 else ""
    except Exception:
        return ""


def main():
    os.chdir("/root")
    incremental = "--incremental" in sys.argv
    max_pages = TOTAL_PAGES
    for arg in sys.argv:
        if arg.startswith("--pages="):
            max_pages = int(arg.split("=")[1])

    # Phase 1
    print("=== Phase 1: Collecting URLs ===")
    if incremental:
        rng = [1]
        print("Mode: INCREMENTAL (page 1)")
    else:
        rng = range(1, min(max_pages, TOTAL_PAGES) + 1)
        print(f"Mode: FULL ({min(max_pages, TOTAL_PAGES)} pages)")

    all_articles = []
    seen = set()

    for p in rng:
        try:
            html = fetch_list_page(p)
            arts = parse_list(html)
            new = 0
            for a in arts:
                if a["url"] not in seen:
                    seen.add(a["url"])
                    all_articles.append(a)
                    new += 1
            print(f"  Page {p}/{TOTAL_PAGES}: {len(arts)} items, +{new}, total={len(all_articles)}")
        except Exception as e:
            print(f"  Page {p} ERROR: {e}")
        time.sleep(0.2)

    print(f"\n=== {len(all_articles)} articles ===")
    if not all_articles:
        return

    with open("/tmp/gzza_articles.json", "w") as f:
        json.dump(all_articles, f, ensure_ascii=False)

    # Phase 2: Details
    print("\n=== Phase 2: Fetching details ===")
    conn = sqlite3.connect(TEMP_DB, timeout=60)
    c = conn.cursor()
    c.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, content TEXT, source_url TEXT UNIQUE,
        site_name TEXT, publish_date TEXT
    )""")
    conn.commit()

    ins, skip, emp = 0, 0, 0
    for i, art in enumerate(all_articles):
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url = ?", (art["url"],))
        if c.fetchone()[0] > 0:
            skip += 1
            continue
        content = fetch_detail(art["url"])
        if content:
            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                    (art["title"], content, art["url"], SITE_NAME, art["date"])
                )
                conn.commit()
                ins += 1
            except Exception as e:
                print(f"  DB error: {e}")
        else:
            emp += 1
        if i % 50 == 0:
            print(f"  [{i}/{len(all_articles)}] ins={ins} skip={skip} emp={emp}")
        time.sleep(0.1)

    conn.close()
    print(f"\n=== Done: ins={ins} skip={skip} emp={emp} ===")


if __name__ == "__main__":
    main()
