#!/usr/bin/env python3
"""
昆山市人民政府-公示公告 (crawl_ksgsgg.py)
ks.gov.cn - UCAP CMS
列表: /kss/gsgg/common_list2.shtml (只有1页有数据)
详情: /kss/gsgg/{YYYYMM}/{uuid}.shtml
"""
import json, os, sys, time, re, urllib.request, sqlite3

SITE_NAME = "ksgsgg"
BASE_URL = "https://www.ks.gov.cn"
LIST_URL = f"{BASE_URL}/kss/gsgg"
TEMP_DB = "/root/temp_search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def fetch_list_page(page_num):
    """Fetch one list page (only page 2 has real data)"""
    url = f"{LIST_URL}/common_list{page_num}.shtml"
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    return resp.read().decode("utf-8", errors="replace")


def parse_list(html):
    """Extract articles from list page - UCAP CMS:
    <h4><a href="/kss/gsgg/{YYYYMM}/{uuid}.shtml" title="title">title</a>
    <span class="time">YYYY-MM-DD</span></h4>
    """
    articles = []
    for m in re.finditer(
        r'<h4><a\s+href="(/kss/gsgg/\d+[^"<>]+\.shtml)"\s+title="([^"]*)"[^>]*>(.*?)</a>\s*<span\s+class="time">\s*(\d{4}-\d{2}-\d{2})\s*</span></h4>',
        html, re.DOTALL
    ):
        href = m.group(1).strip()
        title = m.group(3).strip()  # fallback: use inner text
        if not title:
            title = m.group(2).strip()  # use title attr
        date_str = m.group(4).strip()
        if not href.startswith("http"):
            href = BASE_URL + href
        articles.append({"title": title, "url": href, "date": date_str})
    return articles


def fetch_detail(url):
    """Fetch article detail content - UCAP CMS"""
    try:
        req = urllib.request.Request(url, headers=HEADERS)
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode("utf-8", errors="replace")

        # Update title from meta
        title = ""
        m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
        if m:
            title = m.group(1).strip()

        # Update date from meta
        pub_date = ""
        m = re.search(r'<meta\s+name="PubDate"\s+content="(\d{4}-\d{2}-\d{2})', html)
        if m:
            pub_date = m.group(1)

        # Content: <div class="article-content article-content-body" id="zoomcon"><UCAPCONTENT>...</UCAPCONTENT></div>
        content = ""
        m = re.search(r'<UCAPCONTENT>(.*?)</UCAPCONTENT>', html, re.DOTALL)
        if m:
            content = m.group(1).strip()
        if not content:
            m = re.search(r'id="zoomcon"[^>]*>([\s\S]*?)</div>\s*</div>', html, re.DOTALL)
            if m:
                content = m.group(1).strip()

        if content:
            content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
            content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
            content = re.sub(r'\s*style="[^"]*"', '', content)
            content = re.sub(r'&#13;', '', content)  # remove carriage return entities

        text_len = len(re.sub(r'<[^>]+>', '', content).strip())
        if text_len < 30:
            content = ""

        return title or None, content or None, pub_date or None

    except urllib.error.HTTPError as e:
        return None, None, None if e.code == 404 else None
    except Exception:
        return None, None, None


def main():
    os.chdir("/root")
    incremental = "--incremental" in sys.argv

    # Phase 1: Collect URLs
    print("=== Phase 1: Collecting article URLs ===")
    all_articles = []
    seen_urls = set()

    # Only page 2 has real data for this site
    pages = [2]

    for page in pages:
        try:
            html = fetch_list_page(page)
            articles = parse_list(html)
            new = 0
            for art in articles:
                if art["url"] not in seen_urls:
                    seen_urls.add(art["url"])
                    all_articles.append(art)
                    new += 1
            print(f"  Page {page}: {len(articles)} items, {new} new")
        except Exception as e:
            print(f"  Page {page}: ERROR - {e}")
        time.sleep(0.3)

    total = len(all_articles)
    print(f"\nTotal unique articles: {total}")
    if total == 0:
        print("No articles found. Exiting.")
        return

    # Phase 2: Fetch details
    print("\n=== Phase 2: Fetching details ===")
    conn = sqlite3.connect(TEMP_DB)
    c = conn.cursor()
    c.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        title TEXT, content TEXT, source_url TEXT UNIQUE,
        site_name TEXT, pub_date TEXT
    )""")
    conn.commit()

    inserted = 0
    skipped = 0
    empty = 0

    for i, art in enumerate(all_articles):
        c.execute("SELECT COUNT(*) FROM gov_raw WHERE source_url = ?", (art["url"],))
        if c.fetchone()[0] > 0:
            skipped += 1
            continue

        title, content, pub_date = fetch_detail(art["url"])
        final_title = title or art["title"]
        final_date = pub_date or art["date"]

        if content:
            try:
                c.execute(
                    "INSERT OR IGNORE INTO gov_raw (title, content, source_url, site_name, pub_date) VALUES (?, ?, ?, ?, ?)",
                    (final_title, content, art["url"], SITE_NAME, final_date)
                )
                conn.commit()
                inserted += 1
            except Exception as e:
                print(f"  DB error: {e}")
        else:
            empty += 1

        if i % 10 == 0 or i == total - 1:
            print(f"  [{i}/{total}] ins={inserted} skip={skipped} emp={empty}")

        time.sleep(0.3)

    conn.close()
    print(f"\n=== Done: ins={inserted} skip={skipped} emp={empty} ===")


if __name__ == "__main__":
    main()
