#!/usr/bin/env python3
"""
廉江市政府信息公开平台
========================
CMS: 广东GKML平台 (Vue SPA)
列表API: GET /gkmlpt/api/all/{classify_id}?page={page}&sid={SID}
详情: /gkmlpt/content/{a}/{b}/post_{id}.html
"""
import sys, os, re, urllib.parse, json, warnings, urllib.request
warnings.filterwarnings("ignore")

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

from bs4 import BeautifulSoup

SITE_NAME = "廉江市人民政府-信息公开"
BASE_URL = "http://www.lianjiang.gov.cn"
SID = "4408810028"
API_URL = f"{BASE_URL}/gkmlpt/api/all/0?page={{page}}&sid={SID}"


def fetch_api(page):
    """Fetch list data from GKML API"""
    url = API_URL.format(page=page)
    req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
    try:
        data = json.loads(urllib.request.urlopen(req, timeout=30).read().decode())
        return data.get("articles", []), data.get("total", 0)
    except Exception:
        return [], 0


def fetch_detail(url):
    """Fetch detail page HTML"""
    try:
        req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
        return urllib.request.urlopen(req, timeout=30).read().decode()
    except Exception:
        return None


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_detail(detail_url):
    """Extract content from detail page"""
    html = fetch_detail(detail_url)
    if not html:
        return None, None, None, []
    
    soup = BeautifulSoup(html, "html.parser")
    
    # Title
    title_tag = soup.find("title")
    title = title_tag.get_text(strip=True) if title_tag else ""
    
    # Date from div.date-row
    date_str = None
    date_div = soup.find("div", class_="date-row")
    if date_div:
        dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", date_div.get_text())
        if dm:
            date_str = dm.group(1)
    
    # Content
    body_parts = []
    attachments = []
    
    # Metadata table (first table)
    meta_table = soup.find("table", class_="doc-head") or soup.find("table", style=re.compile(r"margin-bottom"))
    
    # Main content from div.article-content
    ac = soup.find("div", class_="article-content")
    if ac:
        # Extract metadata from the first table (usually has 索引号 etc.)
        tables = ac.find_all("table")
        meta_fields = []
        content_tables = []
        
        for table in tables:
            txt = table.get_text(" ", strip=True)
            if any(k in txt for k in ["索引号", "发布机构", "文号", "分类", "主题分类"]):
                meta_fields.append(table)
            else:
                content_tables.append(table)
        
        # Render metadata
        for table in meta_fields:
            md = html_table_to_html(table, base_url=BASE_URL)
            if md:
                body_parts.append(md)
        
        # Extract text content (<p> elements outside of tables)
        for el in ac.find_all(["p", "div"], recursive=True):
            if el.find_parent("table"):
                continue
            if el.name == "p" or (el.name == "div" and not el.find_all(["p", "table"])):
                # Convert br to newline
                DELIM = "\x00BR\x00"
                for br in el.find_all("br"):
                    br.replace_with(DELIM)
                txt = el.get_text(" ", strip=True)
                txt = txt.replace(DELIM, "\n")
                txt = re.sub(r"\n\s*\n", "\n\n", txt)
                if txt and len(txt) > 10:
                    # Skip short repetitive text
                    body_parts.append(txt)
        
        # Render content tables
        for table in content_tables:
            md = html_table_to_html(table, base_url=BASE_URL)
            if md:
                body_parts.append(md)
        
        # Extract attachments
        for a in ac.find_all("a"):
            href = a.get("href", "")
            fname = a.get_text(strip=True)
            if href and re.search(r"\.(pdf|doc|docx|xls|xlsx|rar|zip)\b", href, re.I):
                full_url = urllib.parse.urljoin(detail_url, href)
                attachments.append({"name": fname, "url": full_url})
    
    body = "\n\n".join(body_parts) if body_parts else None
    return title, date_str, body, attachments


def crawl(pages=5):
    all_items = []
    total = 0
    
    for page_no in range(1, pages + 1):
        articles, total = fetch_api(page_no)
        if not articles:
            break
        all_items.extend(articles)
        print(f"Page {page_no}: {len(articles)} items", flush=True)
    
    print(f"\nTotal list items: {len(all_items)}, total={total}", flush=True)
    
    results = []
    empty_body = 0
    with_attach = 0
    
    for i, item in enumerate(all_items):
        title = item.get("title", "")
        detail_url = item.get("url", "")
        if not detail_url:
            continue
        
        api_date = item.get("date", 0)
        
        parsed_title, pub_date, body, attachments = parse_detail(detail_url)
        final_title = parsed_title or title
        final_date = pub_date or ""
        attach_str = ";".join([f"[{a['name']}]({a['url']})" for a in attachments]) if attachments else ""
        if attachments:
            with_attach += 1
        if not body:
            empty_body += 1
        results.append({
            "site_name": SITE_NAME,
            "title": final_title,
            "url": detail_url,
            "source_url": detail_url,
            "pub_date": final_date,
            "content": body or "",
            "summary": (re.sub(r"\s+", " ", body or "")[:300]) if body else "",
            "attachments": attach_str,
        })
        if (i + 1) % 20 == 0:
            print(f"  Parsed: {i+1}/{len(all_items)}", flush=True)
    
    print(f"Parsed: {len(results)} items ({empty_body} empty body, {with_attach} with attachments)", flush=True)
    push_to_searchdb(results, "lianjiang_gkml")


if __name__ == "__main__":
    pages = int(sys.argv[sys.argv.index("--pages") + 1]) if "--pages" in sys.argv else 5
    crawl(pages=pages)
