#!/usr/bin/env python3
"""
藁城区人民政府 - 公告公示
https://www.gc.gov.cn/columns/f144c2a8-8f30-4b5d-85a7-f8054f1d8a22/index.html
JPAAS CMS - static page 1 only (15 items), full content from detail pages
"""
import urllib.request, re, sys, ssl, json, time
from bs4 import BeautifulSoup

BASE = "https://www.gc.gov.cn"
COL_ID = "f144c2a8-8f30-4b5d-85a7-f8054f1d8a22"
ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE
DB_PATH = "/root/search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
SITE_NAME = "藁城区人民政府-公告公示"

def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30, context=ssl_ctx)
    return resp.read().decode("utf-8", errors="replace")

def extract_items(html):
    items = []
    for m in re.finditer(
        r"<li[^>]*>\s*<a\s+href='([^']+)'[^>]*title='([^']*)'[^>]*>.*?</a>\s*<span[^>]*>(\d{4}-\d{2}-\d{2})</span>\s*</li>",
        html, re.DOTALL
    ):
        href = m.group(1).strip()
        if href.startswith("http"):
            url = href
        elif href.startswith("/"):
            url = BASE + href
        else:
            url = BASE + "/" + href
        items.append({"url": url, "title": m.group(2).strip(), "pub_date": m.group(3).strip()})
    return items

def get_detail(detail_url):
    html = fetch(detail_url)
    if not html:
        return None, None, None, []

    # Title from <title>
    title = None
    m = re.search(r"<title>(.*?)</title>", html, re.DOTALL)
    if m:
        title = m.group(1).strip()
        # Remove site suffix
        title = re.sub(r"\s*[-–—].*$", "", title).strip()

    # Date from meta
    pub_date = None
    m = re.search(r'<meta\s+name=[\'"]PubDate[\'"]\s+content=[\'"](\d{4}-\d{2}-\d{2})', html)
    if m:
        pub_date = m.group(1)

    # Content from id="conN"
    content = ""
    soup = BeautifulSoup(html, "html.parser")
    conN = soup.find(id="conN")
    if conN:
        # Remove scripts, styles
        for t in conN.find_all(["script", "style"]):
            t.decompose()
        # Extract text and images
        parts = []
        for child in conN.children:
            if not child.name:
                txt = str(child).strip()
                if txt:
                    parts.append(txt)
            elif child.name == "img":
                src = child.get("src", "")
                if src:
                    if src.startswith("http"):
                        img_url = src
                    elif src.startswith("/"):
                        img_url = BASE + src
                    else:
                        img_url = detail_url.rsplit("/", 1)[0] + "/" + src
                    alt = child.get("alt", "")
                    parts.append(f"![{alt}]({img_url})")
            else:
                txt = child.get_text(" ", strip=True)
                if txt:
                    parts.append(txt)
        content = "\n\n".join(parts)
        content = re.sub(r"\n{3,}", "\n\n", content)
        # For image-only articles, keep the images
        if not content.strip() or len(content.strip()) < 20:
            img_parts = []
            for img in conN.find_all("img"):
                src = img.get("src", "")
                if src:
                    if src.startswith("http"):
                        img_url = src
                    elif src.startswith("/"):
                        img_url = BASE + src
                    else:
                        img_url = detail_url.rsplit("/", 1)[0] + "/" + src
                    img_parts.append(f"![]({img_url})")
            content = "\n\n".join(img_parts)

    # Attachments
    atts = []
    for m in re.finditer(r'<a\s+[^>]*href="([^"]*\.(?:doc|docx|pdf|xls|xlsx|zip|rar))"[^>]*>([^<]*)</a>', html, re.I):
        href = m.group(1).strip()
        if href.startswith("http"):
            full_url = href
        elif href.startswith("/"):
            full_url = BASE + href
        else:
            full_url = detail_url.rsplit("/", 1)[0] + "/" + href
        atts.append({"name": m.group(2).strip(), "url": full_url})

    return title, pub_date, content, atts

def main():
    print(f"[START] {SITE_NAME}")
    url = f"{BASE}/columns/{COL_ID}/index.html"
    html = fetch(url)
    items = extract_items(html)
    print(f"[PAGE 1] {len(items)} items")

    all_records = []
    empty = 0
    for item in items:
        time.sleep(0.3)
        title, pub_date, content, atts = get_detail(item["url"])
        if title: item["title"] = title
        if pub_date: item["pub_date"] = pub_date
        item["content"] = content or ""
        item["attachments"] = atts
        item["site_name"] = SITE_NAME
        if not content or len(content) < 20:
            empty += 1
        all_records.append(item)
        print(f"  + {item['title'][:30]} | content:{len(content)}B attach:{len(atts)}")

    import sqlite3
    if all_records:
        conn = sqlite3.connect(DB_PATH)
        c = conn.cursor()
        ins = skp = 0
        for rec in all_records:
            try:
                c.execute(
                    """INSERT OR IGNORE INTO gov_raw
                    (page_url, title, content, publish_date, site_name, attachments, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?)""",
                    (rec["url"], rec["title"], rec.get("content", ""),
                     rec.get("pub_date", ""), SITE_NAME,
                     json.dumps(rec.get("attachments", []), ensure_ascii=False),
                     rec.get("pub_date", ""))
                )
                if c.rowcount > 0: ins += 1
                else: skp += 1
            except Exception as e:
                print(f"[DB] {e}", file=sys.stderr)
                skp += 1
        conn.commit()
        conn.close()
        print(f"[DB] Inserted {ins}, skipped {skp}")
    print(f"[DONE] Records={len(all_records)}, Empty={empty}")

if __name__ == "__main__":
    main()
