#!/usr/bin/env python3
"""荥阳市人民政府 - 建设项目环境影响评价（污染防治）"""
import re, sys, os, json, time
import urllib.request, urllib.error
from bs4 import BeautifulSoup
import sqlite3

BASE_URL = "https://public.xingyang.gov.cn"
LIST_URL = "/?a=dir&c=198718&f=198732&page=%d"
SITE_NAME = "荥阳市-建设项目环评"
SITE_DISPLAY = "荥阳市人民政府"
GROUP_NAME = "环评平台"
DB = os.environ.get("SEARCH_DB", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == "--pages" and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=30).read().decode("utf-8", errors="replace")
    except Exception as e:
        print("[WARN] fetch failed: %s - %s" % (url, e), file=sys.stderr)
        return ""

def extract_items(html):
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for a in soup.select("ul.page-list > a.list_item_pubinfo_a"):
        href = a.get("href", "").strip()
        span = a.find("span")
        title = span.get_text(strip=True) if span else a.get_text(strip=True)
        if href and title:
            if href.startswith("/"):
                href = BASE_URL + href
            elif not href.startswith("http"):
                href = BASE_URL + "/" + href.lstrip("/")
            items.append({"title": title, "url": href})
    return items

def parse_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")

    # Title
    title = ""
    ct_title = soup.find("div", class_="content-title")
    if ct_title:
        title = ct_title.get_text(strip=True)
    if not title and soup.title:
        raw = soup.title.string.strip()
        title = re.sub(r"\s*-\s*政务公开\s*$", "", raw)

    # Date from pubinfo-card
    date_text = ""
    pw = soup.find("div", class_="pubinfo-card-wrap")
    if pw:
        for li in pw.find_all("li"):
            icon = li.find("i")
            if icon and icon.get("title") == "发布日期":
                em = li.find("em")
                if em:
                    date_text = em.get_text(strip=True)
        if not date_text:
            for li in pw.find_all("li"):
                icon = li.find("i")
                if icon and "日期" in icon.get("title", ""):
                    em = li.find("em")
                    if em:
                        date_text = em.get_text(strip=True)
                    break

    # Content from content-txt
    content_parts = []
    attachments = []
    ct_div = soup.find("div", class_="content-txt")
    if ct_div:
        for el in ct_div.find_all(["p", "table", "img"], recursive=True):
            if el.name == "img":
                src = el.get("src", "").strip()
                alt = el.get("alt", "").strip()
                if src:
                    if src.startswith("/"):
                        src = BASE_URL + src
                    content_parts.append("![%s](%s)" % (alt, src))
                continue

            # Skip p elements inside table cells (table rendered separately)
            if el.name == "p" and el.find_parent("table"):
                continue

            if el.name == 'table':
                tbl_html = html_table_to_html(el, url)
                if tbl_html:
                    content_parts.append(tbl_html)
            text = el.get_text(strip=True)
            if text:
                content_parts.append(text)

    # Separate pass: find all attachment links
    for a in soup.find_all("a", href=True):
        h = a["href"]
        if "/attachment/" in h.lower() or h.lower().endswith((".pdf", ".doc", ".docx", ".xls", ".xlsx", ".zip", ".rar")):
            ah = a["href"]
            if ah.startswith("/"):
                ah = BASE_URL + ah
            elif not ah.startswith("http"):
                ah = BASE_URL + "/" + ah.lstrip("/")
            attachments.append({"name": a.get_text(strip=True) or ah.split("/")[-1], "url": ah})

    content = "\n\n".join(p for p in content_parts if p.strip())
    return {"title": title, "date": date_text, "content": content, "attachments": attachments}

def main():
    total_new = 0
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()

    all_items = []
    page_num = 1

    while True:
        if _MAX_PG and page_num > _MAX_PG:
            break

        url = BASE_URL + (LIST_URL % page_num)
        html = fetch(url)
        if not html:
            break

        items = extract_items(html)
        if not items:
            print("[INFO] Page %d: no items, stopping" % page_num, file=sys.stderr)
            break

        print("[INFO] Page %d: %d items" % (page_num, len(items)), file=sys.stderr)
        all_items.extend(items)
        page_num += 1

    print("[INFO] Total collected: %d items" % len(all_items), file=sys.stderr)

    for idx, item in enumerate(all_items):
        if not item["url"]:
            continue

        c.execute("SELECT id FROM gov_raw WHERE page_url=?", (item["url"],))
        if c.fetchone():
            continue

        html = fetch(item["url"])
        if not html:
            continue

        detail = parse_detail(html, item["url"])
        title = detail["title"] or item["title"]
        date_text = detail["date"] or ""
        content = detail["content"]
        attachments_json = json.dumps(detail["attachments"], ensure_ascii=False) if detail.get("attachments") else "[]"

        if not content.strip():
            content = title
            print("[WARN] Empty content: %s" % title[:40], file=sys.stderr)

        summary = content[:200].replace("\n", " ").strip()

        try:
            c.execute("INSERT OR IGNORE INTO gov_raw (page_url, title, site_name, group_name, publish_date, summary, content, attachments, date_rank) VALUES (?,?,?,?,?,?,?,?,?)",
                      (item["url"], title, SITE_NAME, GROUP_NAME, date_text, summary, content, attachments_json, date_text))
            conn.commit()
            if c.rowcount:
                total_new += 1
                if total_new % 10 == 0:
                    print("[INFO] Inserted %d/%d" % (total_new, len(all_items)), file=sys.stderr)
        except Exception as e:
            print("[WARN] DB insert failed: %s - %s" % (title[:30], e), file=sys.stderr)

    conn.close()
    print("[DONE] 荥阳: %d new items" % total_new)

if __name__ == "__main__":
    main()
