#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
枝江市人民政府信息公开-行政许可（宜昌市生态环境局枝江市分局）
http://xxgk.zgzhijiang.gov.cn/list.html?depid=189&catid=663
"""
import re, sys, os, json, time, requests

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "*/*",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "http://xxgk.zgzhijiang.gov.cn/list.html?depid=189&catid=663",
}
SITE_NAME = "宜昌市生态环境局枝江市分局-行政许可"
GROUP = "枝江"
PER_PAGE = 20
MAX_PAGES = 5

LIST_API = "http://www.zgzhijiang.gov.cn/show/lists?jsoncallback=jq&areaid=9&webid=189&cid=663&page=%d&pagenums=%d&orderby=0"
DETAIL_API = "http://www.zgzhijiang.gov.cn/show/detail?jsoncallback=jq&areaid=9&id=%s&cache=on"


def fetch_json(url):
    for attempt in range(3):
        try:
            r = requests.get(url, headers=HEADERS, timeout=60)
            text = r.text.strip()
            if not text:
                raise ValueError("Empty response")
            # Check if it's an HTML error page
            if text.startswith("<") or "404" in text[:200]:
                raise ValueError("Got HTML page instead of JSON")
            # Strip JSONP wrapper: callback({...})
            text = re.sub(r'^\s*[^(]*\(', '', text)
            text = re.sub(r'\)\s*$', '', text)
            return json.loads(text)
        except Exception as e:
            if attempt < 2:
                time.sleep(5)
            else:
                raise


def extract_content_from_html(content_html):
    """Parse HTML content into plain text with preserved paragraphs/tables."""
    parts = []
    attachments = []

    # Extract attachment links
    for a_tag in re.finditer(r'<a[^>]*href="([^"]*\.(?:pdf|doc|docx|xls|xlsx|rar|zip))"[^>]*>(.*?)</a>', content_html, re.I | re.DOTALL):
        link_url = a_tag.group(1)
        link_text = re.sub(r"<[^>]+>", "", a_tag.group(2)).strip()
        attachments.append({"name": link_text or link_url.split("/")[-1], "url": link_url})

    # Extract images
    for img_url in re.findall(r'<img[^>]*src="([^"]+)"', content_html):
        attachments.append({"name": img_url.split("/")[-1], "url": img_url})

    # Parse paragraphs and tables
    for elem in re.finditer(r'<p[^>]*>(.*?)</p>|<table[^>]*>(.*?)</table>', content_html, re.DOTALL | re.I):
        tag = elem.group(0)
        if tag.group(0).startswith('<table'):
            parts.append(tag.group(0))
        else:
            p_text = elem.group(1)
            p_text = re.sub(r'<[^>]+>', '', p_text)
            p_text = re.sub(r'&[nN][bB][sS][pP];', ' ', p_text)
            p_text = p_text.strip()
            if p_text:
                parts.append(p_text)

    content = "\n\n".join(parts)

    # Fallback
    if not content or len(content.strip()) < 20:
        text = re.sub(r'<[^>]+>', ' ', content_html)
        text = re.sub(r'\s+', ' ', text).strip()
        if text and len(text) > 20:
            content = text

    return content, attachments


def fetch_list_page(page):
    """Fetch list items from API."""
    url = LIST_API % (page, PER_PAGE)
    data = fetch_json(url)
    items = data.get("lists", [])
    print("  Page %d: %d items" % (page, len(items)))
    return items


def parse_list_item(item):
    """Parse a list API item."""
    url = "http://xxgk.zgzhijiang.gov.cn/show.html?aid=9&id=%s" % item["n_id"]
    title = item.get("title", "").strip()
    date_str = item.get("vc_inputtime", "").strip()
    if date_str and len(date_str) > 10:
        date_str = date_str[:10]
    return {
        "url": url,
        "title": title,
        "date": date_str,
        "n_id": item["n_id"],
        "api_url": DETAIL_API % item["n_id"],
    }


def fetch_detail(api_url):
    """Fetch detail via API."""
    data = fetch_json(api_url)
    if isinstance(data, list) and len(data) > 0:
        record = data[0]
    elif isinstance(data, dict):
        record = data
    else:
        return None

    title = record.get("title", "").strip()
    pub_date = record.get("vc_inputtime", "")
    if pub_date and len(pub_date) > 10:
        pub_date = pub_date[:10]

    content_html = record.get("content", "")
    content, attachments = extract_content_from_html(content_html)

    # Add fujian (attachment file)  
    fujian = record.get("fujian", "")
    if fujian and fujian.strip() and fujian != "0":
        attachments.append({"name": fujian.split("/")[-1], "url": fujian})

    return {
        "title": title,
        "content": content,
        "date": pub_date,
        "attachments": attachments,
        "raw_title": record.get("title", ""),
        "raw_content": content_html,
    }


def crawl(test_mode=False, max_pages=MAX_PAGES):
    import sqlite3

    all_items = []
    seen_ids = set()

    for page in range(1, max_pages + 1):
        try:
            data = fetch_list_page(page)
        except Exception as e:
            print("  Error on page %d: %s" % (page, e))
            continue

        for item in data:
            n_id = item.get("n_id", "")
            if n_id in seen_ids:
                continue
            seen_ids.add(n_id)
            parsed = parse_list_item(item)
            all_items.append(parsed)

        if test_mode and len(all_items) >= 3:
            break

    print("\nTotal unique items: %d" % len(all_items))

    if test_mode:
        for item in all_items[:3]:
            print("\n=== %s ===" % item["title"][:40])
            result = fetch_detail(item["api_url"])
            if result:
                print("  Title: %s" % result["title"])
                print("  Date: %s" % result["date"])
                print("  Content (%d chars): %s" % (len(result["content"]), result["content"][:200]))
                print("  Attachments: %d" % len(result["attachments"]))
                for att in result["attachments"][:3]:
                    print("    %s -> %s" % (att["name"], att["url"][:60]))
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=30000")
    c = conn.cursor()

    inserted = 0
    existing = 0
    total = len(all_items)

    for idx, item in enumerate(all_items):
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
        if c.fetchone():
            existing += 1
            continue

        try:
            result = fetch_detail(item["api_url"])
        except Exception as e:
            print("  Error fetching %s: %s" % (item["n_id"], e))
            existing += 1
            continue

        if not result:
            existing += 1
            continue

        attachments_json = json.dumps(result["attachments"], ensure_ascii=False) if result["attachments"] else ""

        c.execute(
            """INSERT INTO gov_raw (title, content, summary, publish_date, page_url, site_name, attachments, group_name, source_url)
               VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
            (
                result["title"],
                result["content"],
                (result["title"] + " " + SITE_NAME + " " + (result["content"][:200] if result["content"] else "")),
                result["date"] or item["date"],
                item["url"],
                SITE_NAME,
                attachments_json,
                GROUP,
                item["api_url"],
            ),
        )
        inserted += 1
        if inserted % 10 == 0:
            conn.commit()
            print("  Progress: %d/%d inserted..." % (inserted, total))

    conn.commit()
    conn.close()
    print("\nDone. Inserted: %d, Existing: %d (total: %d)" % (inserted, existing, total))


if __name__ == "__main__":
    if "--test" in sys.argv:
        crawl(test_mode=True)
    else:
        crawl(test_mode=False, max_pages=MAX_PAGES)
