#!/usr/bin/env python3
"""
宿迁市生态环境局 - 建设项目信息公开
https://sthj.suqian.gov.cn/shbj/jsxm/xxgk_list.shtml
Pagination: static (createPageHTML), 173 pages, 16/page
Detail: UCAPCONTENT content, meta ArticleTitle, meta PubDate
"""

import urllib.request, json, re, sys, ssl, time
from bs4 import BeautifulSoup

BASE_URL = "https://sthj.suqian.gov.cn"
LIST_PATH = "/shbj/jsxm/xxgk_list"
TOTAL_PAGES = 173

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE
DB_PATH = "/root/search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def fetch_url(url, timeout=30):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=timeout, context=ssl_ctx)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        print(f"[WARN] {url[:60]}: {e}", file=sys.stderr)
        return None


def extract_list_items(html):
    items = []
    m = re.search(r'<ul\s+class="listContent">(.*?)</ul>', html, re.DOTALL)
    if not m:
        return items

    # Find all <li> blocks
    lis = re.findall(r'<li>(.*?)</li>', m.group(1), re.DOTALL)
    for li in lis:
        # Extract href, title (may be single or double quoted), and date
        am = re.search(r'<a\s+href="([^"]*)"', li)
        if not am:
            continue

        href = am.group(1).strip()
        if href.startswith("/"):
            url = BASE_URL + href
        elif not href.startswith("http"):
            url = BASE_URL + "/shbj/jsxm/" + href
        else:
            url = href

        # Title - try double quotes first, then single
        title = None
        tm = re.search(r'title="([^"]*)"', li)
        if tm:
            title = tm.group(1).strip()
        else:
            tm = re.search(r"title='([^']*)'", li)
            if tm:
                title = tm.group(1).strip()

        # Date
        dm = re.search(r'<span>(\d{4}-\d{2}-\d{2})</span>', li)
        publish_date = dm.group(1).strip() if dm else ""

        if title and url:
            items.append({"url": url, "title": title, "publish_date": publish_date})

    return items


def fetch_detail(detail_url):
    html = fetch_url(detail_url)
    if not html:
        return None, None, None, []

    title = None
    m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
    if m:
        title = re.sub(r"\s+", " ", m.group(1)).strip()

    publish_date = None
    m = re.search(r'<meta\s+name="PubDate"\s+content="(\d{4}-\d{2}-\d{2})', html)
    if m:
        publish_date = m.group(1)

    content = None
    m = re.search(r"<UCAPCONTENT>(.*?)</UCAPCONTENT>", html, re.DOTALL)
    if m:
        content = extract_content(m.group(1), detail_url)

    attachments = extract_attachments(html, detail_url)
    return title, publish_date, content, attachments


def resolve_url(page_url, src):
    if src.startswith("http"):
        return src
    if src.startswith("/"):
        return BASE_URL + src
    return page_url.rsplit("/", 1)[0] + "/" + src


def extract_content(raw, page_url):
    soup = BeautifulSoup(raw, "html.parser")
    for t in soup.find_all(["script", "style"]):
        t.decompose()

    parts = []
    for child in soup.children:
        if not child.name:
            continue
        if child.name == "table" and len(child.get_text(strip=True)) >= 15:
            parts.append(str(child))
        elif child.name == "img":
            src = child.get("src", "")
            if src:
                parts.append("![](" + resolve_url(page_url, src) + ")")
        elif child.name in ["p", "div", "span", "section"]:
            text = child.get_text(" ", strip=True)
            if text:
                parts.append(text)

    result = "\n\n".join(parts)
    result = re.sub(r"\n{3,}", "\n\n", result)
    return result


def extract_attachments(html, page_url):
    atts = []
    p = re.compile(
        r'<a\s+[^>]*href="([^"]*\.(?:doc|docx|pdf|xls|xlsx|zip|rar))"[^>]*>([^<]*)</a>', re.I
    )
    for m in p.finditer(html):
        atts.append({"name": m.group(2).strip(), "url": resolve_url(page_url, m.group(1).strip())})
    return atts


def push_to_searchdb(records):
    import sqlite3
    if not records:
        print("[SKIP] No records")
        return
    conn = sqlite3.connect(DB_PATH)
    c = conn.cursor()
    ins = skp = 0
    for rec in records:
        try:
            c.execute(
                """INSERT OR IGNORE INTO gov_raw
                (page_url, title, content, publish_date, site_name, attachments, date_rank)
                VALUES (?, ?, ?, ?, ?, ?, ?)""",
                (
                    rec["url"],
                    rec["title"],
                    rec.get("content", ""),
                    rec.get("publish_date", ""),
                    "宿迁市生态环境局-建设项目",
                    json.dumps(rec.get("attachments", []), ensure_ascii=False),
                    rec.get("publish_date", ""),
                ),
            )
            if c.rowcount > 0:
                ins += 1
            else:
                skp += 1
        except Exception as e:
            print(f"[DB] Error: {e}", file=sys.stderr)
            skp += 1
    conn.commit()
    conn.close()
    print(f"[DB] Inserted {ins}, skipped {skp}")


def main():
    args = sys.argv[1:]
    max_pages = TOTAL_PAGES
    if args and args[0].isdigit():
        max_pages = int(args[0])

    print(f"[START] pages={max_pages}")
    all_records = []
    empty = total = 0
    last_page = 0

    for page in range(1, max_pages + 1):
        list_url = f"{LIST_PATH}.shtml" if page == 1 else f"{LIST_PATH}_{page}.shtml"
        html = fetch_url(BASE_URL + list_url)
        if not html:
            print(f"[PAGE {page}] Failed")
            continue

        items = extract_list_items(html)
        if not items:
            if page > 1:
                print(f"[PAGE {page}] No items (end)")
                break
            print(f"[PAGE {page}] No items")
            continue

        last_page = page
        total += len(items)
        print(f"[PAGE {page}] {len(items)} items", end="")

        for item in items:
            time.sleep(0.3)
            title, publish_date, content, attachments = fetch_detail(item["url"])
            if title:
                item["title"] = title
            if publish_date:
                item["publish_date"] = publish_date
            item["content"] = content or ""
            item["attachments"] = attachments
            if not content or len(content.strip()) < 20:
                empty += 1
            all_records.append(item)

        print(f" | total: {len(all_records)}")

    print(f"\n[SUMMARY] Pages={last_page} Items={total} Records={len(all_records)} Empty={empty}")
    push_to_searchdb(all_records)
    print("[DONE]")


if __name__ == "__main__":
    main()
