#!/usr/bin/env python3
"""
龙口市人民政府 — 建设项目环评 (col/col48329)
=============================================
JCMS (Hanweb) with DBApp WAF (self-signed cert).
列表: API GET /api-gateway/jpaas-publish-server/front/page/build/unit
详情: /col/col48329/art/2026/art_{hash}.html (content in div#zoom)
"""
import sys, os, re, json, urllib.parse, time, warnings
warnings.filterwarnings("ignore")

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

import requests
from bs4 import BeautifulSoup

SITE_NAME = "龙口市-建设项目环评"
BASE_DOMAIN = "www.longkou.gov.cn"
CDN_IP = "101.206.185.117"
LIST_URL = f"https://{CDN_IP}/api-gateway/jpaas-publish-server/front/page/build/unit"
DETAIL_BASE = f"https://{CDN_IP}"

API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "153",
    "tplSetId": "U1S5IC8QNYd71CkF9ENIL",
    "pageType": "column",
    "tagId": "文章列表",
    "editType": "null",
    "pageId": "48329",
}
PAGE_SIZE = 15
HEADERS = {
    "Host": BASE_DOMAIN,
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
}


def fetch(url, retries=3):
    for attempt in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, verify=False, timeout=30)
            r.encoding = 'utf-8'
            return r.text
        except Exception as e:
            if attempt < retries - 1:
                time.sleep(2)
            else:
                print(f"  [ERROR] {url[:80]}: {e}", file=sys.stderr)
                return None


def fetch_list_page(page_no):
    """Fetch one list page from JCMS unitbuild API"""
    params = dict(API_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": PAGE_SIZE}, ensure_ascii=False)
    url = LIST_URL + "?" + urllib.parse.urlencode(params)
    text = fetch(url)
    if not text:
        return [], 0
    try:
        data = json.loads(text)
    except json.JSONDecodeError:
        return [], 0
    if not data.get("success"):
        return [], 0
    html = data["data"]["html"]
    count_m = re.search(r'count="(\d+)"', html)
    total = int(count_m.group(1)) if count_m else 0
    # Extract items: <a href="..." title="..." >text</a><span>date</span>
    items = []
    pattern = re.compile(
        r'<a\s+href="([^"]+)"[^>]*title="([^"]*)"[^>]*>(.*?)</a>\s*<span>([^<]*)</span>'
    )
    for m in pattern.finditer(html):
        detail_url = urllib.parse.urljoin(DETAIL_BASE, m.group(1))
        title = m.group(2).strip()
        date_str = m.group(4).strip()
        items.append({"title": title, "url": detail_url, "date": date_str})
    return items, total


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_detail(detail_url):
    """Extract content from detail page"""
    text = fetch(detail_url)
    if not text:
        return None, None, None, []
    soup = BeautifulSoup(text, 'html.parser')
    # Title from <title>
    title_tag = soup.find('title')
    title = title_tag.get_text(strip=True) if title_tag else ""
    title = re.sub(r'^[^ ]+ ', '', title)  # remove site prefix
    # Content container
    zoom = soup.find(id='zoom')
    if not zoom:
        return title, None, None, []
    # Publish date
    date_str = None
    date_m = re.search(r'发布日期[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', text)
    if date_m:
        date_str = date_m.group(1)
    # Extract body: iterate p, table, img in order
    body_parts = []
    attachments = []
    for el in zoom.find_all(['p', 'table', 'img'], recursive=False):
        if el.name == 'p':
            txt = el.get_text(" ", strip=True)
            if txt:
                body_parts.append(txt)
        elif el.name == 'table':
            md = html_table_to_html(el)
            if md:
                body_parts.append(md)
        elif el.name == 'img':
            src = el.get('src', '')
            alt = el.get('alt', '')
            if src:
                full_src = urllib.parse.urljoin(DETAIL_BASE, src)
                body_parts.append(f"![{alt}]({full_src})")
    # If body is very short, try recursive walk inside div wrappers
    body_text = "\n\n".join(body_parts)
    if len(body_text) < 100:
        # Try deeper: find all p/table/img including nested
        body_parts = []
        for el in zoom.find_all(['p', 'table', 'img']):
            # Skip tables that are inside <p> (invalid but possible)
            if el.find_parent('table') and el.name != 'table':
                continue
            if el.name == 'p':
                # Skip if this p is inside a table
                if el.find_parent('table'):
                    continue
                txt = el.get_text(" ", strip=True)
                if txt:
                    body_parts.append(txt)
            elif el.name == 'table':
                md = html_table_to_html(el)
                if md:
                    body_parts.append(md)
            elif el.name == 'img':
                src = el.get('src', '')
                alt = el.get('alt', '')
                if src:
                    full_src = urllib.parse.urljoin(DETAIL_BASE, src)
                    body_parts.append(f"![{alt}]({full_src})")
        body_text = "\n\n".join(body_parts)
    # Attachments: <a data-link="...">filename</a>
    for a in zoom.find_all('a'):
        data_link = a.get('data-link')
        if data_link:
            fname = a.get_text(strip=True)
            full_link = urllib.parse.urljoin(DETAIL_BASE, data_link)
            attachments.append({"name": fname, "url": full_link})
    return title, date_str, body_text or None, attachments


def crawl(pages=5):
    all_items = []
    total_count = 0
    for page_no in range(1, pages + 1):
        items, total = fetch_list_page(page_no)
        if not items:
            break
        total_count = total
        all_items.extend(items)
        print(f"Page {page_no}: {len(items)} items", flush=True)
        if len(all_items) >= total_count:
            break
    print(f"\nTotal list items: {len(all_items)}", flush=True)
    results = []
    empty_body = 0
    with_attach = 0
    for i, item in enumerate(all_items):
        title = item["title"]
        detail_url = item["url"]
        list_date = item["date"]
        parsed_title, pub_date, body, attachments = parse_detail(detail_url)
        final_title = parsed_title or title
        final_date = pub_date or list_date
        attach_str = ";".join([a["name"] for a in attachments]) if attachments else ""
        if attachments:
            with_attach += 1
        if not body:
            empty_body += 1
        results.append({
            "site_name": SITE_NAME,
            "title": final_title,
            "url": detail_url,
            "source_url": detail_url,
            "pub_date": final_date,
            "content": body or "",
            "summary": (re.sub(r'\s+', ' ', body or "")[:300]) if body else "",
            "attachments": attach_str,
        })
        if (i + 1) % 10 == 0:
            print(f"  Parsed: {i+1}/{len(all_items)}", flush=True)
    print(f"Parsed: {len(results)} items ({empty_body} empty body, {with_attach} with attachments)", flush=True)
    push_to_searchdb(results, "longkou_hjhp")


if __name__ == "__main__":
    pages = int(sys.argv[sys.argv.index("--pages") + 1]) if "--pages" in sys.argv else 5
    crawl(pages=pages)
