#!/usr/bin/env python3
"""
南陵县生态环境分局 - 行政审批 (crawl_nlx_xzsp.py)
芜湖市政府网站群CMS
API: GET /wuhu/site/label/8888?labelName=publicInfoList (列表)
     GET /wuhu/site/label/8888?labelName=publicContentDetail&isJson=true (详情)
"""
import json, os, sys, time, re, urllib.request

SITE_NAME = "nlx_xzsp"
BASE_URL = "https://www.nlx.gov.cn"
LABEL_URL = f"{BASE_URL}/wuhu/site/label/8888"
TEMP_DB = "/root/search.db"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": f"{BASE_URL}/public/column/6605271?type=4&catId=7092673&action=list"
}

ORGAN_ID = "6605271"
CAT_ID = "7092673"
SITE_ID = "6787891"


def fetch_list(page_index):
    """Fetch one page of article list"""
    params = {
        "labelName": "publicInfoList",
        "siteId": SITE_ID,
        "organId": ORGAN_ID,
        "pageSize": 20,
        "pageIndex": page_index,
        "isDate": "true",
        "dateFormat": "yyyy-MM-dd",
        "length": 50,
        "type": 4,
        "action": "list",
        "isJson": "true",
        "isRel": "true",
        "catId": CAT_ID,
        "catalogType": ""
    }
    qs = "&".join(f"{k}={urllib.request.quote(str(v))}" if not isinstance(v, str) else f"{k}={v}"
                   for k, v in params.items())
    req = urllib.request.Request(f"{LABEL_URL}?{qs}", headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    return json.loads(resp.read().decode("utf-8"))


def fetch_detail(content_id):
    """Fetch article detail"""
    params = {
        "labelName": "publicContentDetail",
        "contentId": content_id,
        "isJson": "true"
    }
    qs = "&".join(f"{k}={v}" for k, v in params.items())
    try:
        req = urllib.request.Request(f"{LABEL_URL}?{qs}", headers=HEADERS)
        resp = urllib.request.urlopen(req, timeout=30)
        return json.loads(resp.read().decode("utf-8"))
    except Exception as e:
        print(f"    Detail error for {content_id}: {e}")
        return None


def clean_content(html):
    """Clean up HTML content"""
    if not html:
        return ""
    html = re.sub(r'<script[^>]*>.*?</script>', '', html, flags=re.DOTALL)
    html = re.sub(r'<style[^>]*>.*?</style>', '', html, flags=re.DOTALL)
    text_len = len(re.sub(r'<[^>]+>', '', html).strip())
    if text_len < 30:
        return ""
    # Fix relative URLs
    html = re.sub(r'(src|href)=(["\'])(/[^"\']+)',
                  lambda m: f'{m.group(1)}={m.group(2)}{BASE_URL}{m.group(3)}', html)
    return html


def main():
    os.chdir("/root")
    incremental = "--incremental" in sys.argv
    if incremental:
        print("Mode: INCREMENTAL (page 1 only)")
    else:
        print("Mode: FULL")

    # Phase 1: Collect articles
    print("=== Phase 1: Collecting articles ===")
    all_articles = []
    total_pages = 99  # Will be corrected after first page

    for page in range(1, total_pages + 1):
        try:
            result = fetch_list(page)
            items = result.get("data", [])
            total = result.get("total", 0)
            per_page = result.get("pageSize", 20)
            total_pages = (total + per_page - 1) // per_page

            for item in items:
                all_articles.append({
                    "content_id": item["contentId"],
                    "title": item["title"],
                    "publish_date": item.get("publishDate", ""),
                    "link": item.get("link", "")
                })

            print(f"  Page {page}/{total_pages}: {len(items)} items, {len(all_articles)} total")

            if page >= total_pages:
                break
            if incremental:
                break
            time.sleep(0.3)

        except Exception as e:
            print(f"  Page {page} ERROR: {e}")
            break

    print(f"\n=== Collected {len(all_articles)} articles ===")

    # Phase 2: Fetch details and save
    print("\n=== Phase 2: Fetching details & saving ===")
    import sqlite3
    conn = sqlite3.connect(TEMP_DB, timeout=60)
    c = conn.cursor()
    c.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, content TEXT, source_url TEXT UNIQUE,
        site_name TEXT, publish_date TEXT
    )""")
    conn.commit()

    inserted = 0
    skipped = 0
    empty = 0

    for i, art in enumerate(all_articles):
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url = ?", (art["link"],))
        if c.fetchone()[0] > 0:
            skipped += 1
            continue

        # Fetch detail
        detail = fetch_detail(art["content_id"])
        if detail and detail.get("content"):
            content = clean_content(detail["content"])
            if content:
                try:
                    c.execute(
                        "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, publish_date) VALUES (?, ?, ?, ?, ?)",
                        (art["title"], content, art["link"], SITE_NAME, art["publish_date"][:10])
                    )
                    conn.commit()
                    inserted += 1
                except Exception as e:
                    print(f"    DB error: {e}")
            else:
                empty += 1
        else:
            empty += 1

        if i % 10 == 0:
            print(f"  [{i}/{len(all_articles)}] inserted={inserted} skipped={skipped} empty={empty}")

    conn.close()
    print(f"\n=== Done ===")
    print(f"Total: {len(all_articles)}")
    print(f"Inserted: {inserted}")
    print(f"Skipped: {skipped}")
    print(f"Empty content: {empty}")


if __name__ == "__main__":
    main()
