#!/usr/bin/env python3
"""
怀远县人民政府 - 建设项目环评审批 (crawl_ahhy_hpsp.py)
蚌埠市信息公开系统 - site/label/8888 API
总计398条, 27页(15条/页)
"""
import json, os, sys, time, re, urllib.request, sqlite3

SITE_NAME = "ahhy_hpsp"
BASE_URL = "https://www.ahhy.gov.cn"
API_URL = f"{BASE_URL}/zfxxgk/site/label/8888"
TEMP_DB = "/root/search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

TOTAL_PAGES = 27
PAGE_SIZE = 15


def fetch_list_page(page_index):
    """Fetch one page from label API"""
    params = urllib.parse.urlencode({
        "labelName": "publicDynamicInfoList",
        "siteId": "6795621",
        "organId": "29621",
        "catId": "18193621",
        "pageIndex": page_index,
        "pageSize": PAGE_SIZE,
        "isJson": "true"
    })
    url = f"{API_URL}?{params}"
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    return json.loads(resp.read().decode("utf-8"))


def parse_list(data):
    """Extract articles from API response"""
    articles = []
    result = data.get("resultList", {})
    for item in result.get("data", []):
        title = item.get("title", "").strip()
        content_id = item.get("contentId", "")
        date = item.get("publishDate") or item.get("createDate", "")
        if date:
            date = date[:10]
        if title and content_id:
            url = f"{BASE_URL}/zfxxgk/public/content/{content_id}"
            articles.append({"title": title, "url": url, "date": date})
    return articles


def fetch_detail(url):
    """Fetch article detail content"""
    try:
        req = urllib.request.Request(url, headers=HEADERS)
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode("utf-8", errors="replace")

        # Title from meta
        title = ""
        m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
        if m:
            title = m.group(1).strip()

        # Content from id="zoom"
        content = ""
        m = re.search(r'id="zoom"[^>]*>([\s\S]*?)</div>\s*</div>', html, re.DOTALL | re.I)
        if not m:
            m = re.search(r'id="zoom"[^>]*>([\s\S]*?)</div>', html, re.DOTALL | re.I)
        if m:
            content = m.group(1).strip()

        if not content:
            m = re.search(r'id="Zoom"[^>]*>([\s\S]*?)</div>', html, re.DOTALL | re.I)
            if m:
                content = m.group(1).strip()

        if content:
            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)

        text_len = len(re.sub(r'<[^>]+>', '', content).strip())
        return (title or None, content or None) if text_len >= 30 else (title or None, None)

    except urllib.error.HTTPError as e:
        return None, None if e.code == 404 else None
    except Exception:
        return None, None


def main():
    os.chdir("/root")
    incremental = "--incremental" in sys.argv

    # Phase 1: Collect URLs
    print("=== Phase 1: Collecting article URLs ===")
    all_articles = []
    seen_urls = set()

    pages_to_fetch = [1] if incremental else range(1, TOTAL_PAGES + 1)
    mode = "INCREMENTAL" if incremental else f"FULL ({TOTAL_PAGES} pages)"
    print(f"Mode: {mode}")

    for page in pages_to_fetch:
        try:
            data = fetch_list_page(page)
            articles = parse_list(data)
            new = 0
            for art in articles:
                if art["url"] not in seen_urls:
                    seen_urls.add(art["url"])
                    all_articles.append(art)
                    new += 1
            print(f"  Page {page}: {len(articles)} items, {new} new", flush=True)
        except Exception as e:
            print(f"  Page {page}: ERROR - {e}", flush=True)
        time.sleep(0.3)

    total = len(all_articles)
    print(f"\nTotal unique articles: {total}")
    if total == 0:
        print("No articles found. Exiting.")
        return

    # Phase 2: Fetch details
    print("\n=== Phase 2: Fetching details ===")
    conn = sqlite3.connect(TEMP_DB, timeout=60)
    c = conn.cursor()
    c.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, content TEXT, source_url TEXT UNIQUE,
        site_name TEXT, publish_date TEXT
    )""")
    conn.commit()

    inserted = 0
    skipped = 0
    empty = 0

    for i, art in enumerate(all_articles):
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url = ?", (art["url"],))
        if c.fetchone()[0] > 0:
            skipped += 1
            continue

        detail_title, content = fetch_detail(art["url"])
        final_title = detail_title or art["title"]

        if content:
            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                    (final_title, content, art["url"], SITE_NAME, art["date"])
                )
                conn.commit()
                inserted += 1
            except Exception as e:
                print(f"  DB error: {e}")
        else:
            empty += 1

        if i % 20 == 0 or i == total - 1:
            print(f"  [{i}/{total}] ins={inserted} skip={skipped} emp={empty}", flush=True)

        time.sleep(0.2)

    conn.close()
    print(f"\n=== Done: ins={inserted} skip={skipped} emp={empty} ===")


if __name__ == "__main__":
    main()
