#!/usr/bin/env python3
"""永登县人民政府 - 法定主动公开内容 (col16682)"""
import re, sys, os, json, html as html_mod
import urllib.request, urllib.error
from bs4 import BeautifulSoup
import sqlite3

BASE_URL = "https://www.yongdeng.gov.cn"
LIST_URL = BASE_URL + "/col/col16682/index.html"
SITE_NAME = "永登县-法定主动公开内容"
SITE_DISPLAY = "永登县人民政府"
DB = os.environ.get("SEARCH_DB", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == "--pages" and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    return urllib.request.urlopen(req, timeout=30).read().decode()

def extract_items_from_page(html):
    """Extract items from the datastore CDATA"""
    items = []
    # Find all records in the datastore
    for m in re.finditer(r'<record><!\[CDATA\[(.*?)\]\]></record>', html, re.DOTALL):
        record_html = m.group(1)
        # Parse the li element
        soup = BeautifulSoup(record_html, "html.parser")
        li = soup.find("li")
        if not li:
            continue
        a = li.find("a", href=True) if li else None
        if not a:
            continue
        href = a["href"].strip()
        if href.startswith("/"):
            href = BASE_URL + href
        title = a.get("title", "") or a.get_text(" ", strip=True)
        # Date from <b> tag
        b = li.find("b")
        date_str = b.get_text(" ", strip=True) if b else ""
        items.append({"title": title, "url": href, "date": date_str})
    return items

def parse_detail(html):
    soup = BeautifulSoup(html, "html.parser")
    
    # Title
    title = ""
    if soup.title:
        t = soup.title.string.strip()
        # Remove site suffix
        title = re.sub(r"永登县人民政府\s+法定主动公开内容\s*", "", t).strip()
    if not title:
        t = soup.find("div", class_="zfxxgk_con")
        if t:
            lines = t.get_text(" ", strip=True).split("发布日期")[0].strip()
            title = lines[:200] if lines else ""
    
    # Date
    date_text = ""
    con = soup.find("div", class_="zfxxgk_con")
    if con:
        m = re.search(r"发布日期[：:](\d{4}-\d{2}-\d{2})", con.get_text())
        if m:
            date_text = m.group(1)
    
    # Content
    content_div = soup.find("div", class_="zfxxgk_con")
    if not content_div:
        return {"title": title, "date": date_text, "content": "", "attachments": []}
    
    parts = []
    attachments = []
    
    for child in content_div.find_all(["p", "table", "img"], recursive=True):
        if child.name == "p":
            if child.find_parent("table"):
                continue
            text = child.get_text(" ", strip=True)
            a_tag = child.find("a", href=True)
            if a_tag:
                ahref = a_tag["href"]
                if ahref.startswith("/"):
                    ahref = BASE_URL + ahref
                if any(ahref.lower().endswith(ext) for ext in [".pdf", ".doc", ".docx", ".xls", ".xlsx"]):
                    attachments.append({"name": a_tag.get_text(" ", strip=True) or ahref.split("/")[-1], "url": ahref})
            if text:
                # Skip breadcrumb/UI text
                if any(k in text for k in ["当前位置", "首页", "字号", "打印", "关闭"]):
                    continue
                parts.append(text)
        elif child.name == 'table':
            tbl_html = html_table_to_html(child, BASE_URL)
            if tbl_html:
                parts.append(tbl_html)
        elif child.name == "img":
            src = child.get("src", "")
            alt = child.get("alt", "")
            if src:
                if src.startswith("/"):
                    src = BASE_URL + src
                parts.append("![" + (alt or "image") + "](" + src + ")")
    
    content = "\n\n".join(parts)
    return {"title": title, "date": date_text, "content": content, "attachments": attachments}

def main():
    conn = sqlite3.connect(DB)
    c = conn.cursor()
    
    html = fetch(LIST_URL)
    all_items = extract_items_from_page(html)
    total_count = len(all_items)
    print("Total items in datastore:", total_count)
    
    pages_to_fetch = _MAX_PG if _MAX_PG else 99  # _MAX_PG limits detail fetches, not pages
    
    existing = set()
    for row in c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        existing.add(row[0])
    print("Existing:", len(existing))
    
    new_count = 0
    skip_count = 0
    empty_count = 0
    
    for idx, item in enumerate(all_items):
        url = item["url"]
        if url in existing:
            skip_count += 1
            continue
        
        # Limit by page count (not page number but item count)
        if new_count >= pages_to_fetch * 20:
            print("  Reached max items limit")
            break
        
        try:
            html = fetch(url)
            detail = parse_detail(html)
            
            title = detail["title"] or item["title"]
            date_text = detail["date"] or item["date"]
            content = detail["content"]
            attachments = detail["attachments"]
            
            if not content:
                empty_count += 1
                print("  WARN empty:", title[:30])
            
            attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else "[]"
            text_only = re.sub(r"\s+", " ", content)
            text_only = re.sub(r"!\[.*?\]\(.*?\)", "", text_only)
            text_only = re.sub(r"\|.*?\|", "", text_only)
            summary = (text_only[:200] or title).strip()
            
            c.execute(
                "INSERT OR REPLACE INTO gov_raw (title, site_name, page_url, publish_date, content, summary, attachments, date_rank) VALUES (?,?,?,?,?,?,?,?)",
                (title, SITE_NAME, url, date_text, content, summary, attachments_json, date_text)
            )
            conn.commit()
            new_count += 1
            print("  OK", title[:30], "|", date_text)
        except Exception as e:
            print("  ERR", item["title"][:30], "|", e)
    
    print("\n" + "=" * 40)
    print(SITE_DISPLAY, "(", SITE_NAME, ")")
    print("New:", new_count, "Skip:", skip_count, "Empty:", empty_count)
    conn.close()

if __name__ == "__main__":
    main()
