#!/usr/bin/env python3
"""
蓬莱区-建设项目环评公示
API: /api-gateway/jpaas-publish-server/front/page/build/unit (GET)
列表: AJAX加载，#当前栏目列表 > .page-content > li > a[title] + span(YYYY-MM-DD)
分页: paramJson={"pageNo":N, "pageSize":25}, 1125条/45页
详情: div#zoom > p, @href .doc/.pdf 附件
"""

import requests, re, sqlite3, time, os, json as _json, sys
from bs4 import BeautifulSoup
from requests.adapters import HTTPAdapter
from urllib3.util.retry import Retry

API_URL = "https://www.penglai.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
LIST_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "152",
    "tplSetId": "awU93spIZ5G7CYnk12KGt",
    "pageType": "column",
    "tagId": "当前栏目列表",
    "editType": "null",
    "pageId": "30455",
}
SITE_NAME = "蓬莱区-建设项目环评"
DB_PATH = "/root/search.db"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
BASE_URL = "https://www.penglai.gov.cn"
MAX_PAGES = 5  # 默认前5页（最新约125条），日跑足够

session = requests.Session()
session.headers.update(HEADERS)


def fetch_list_page(page_no):
    """通过API获取列表页"""
    params = dict(LIST_PARAMS)
    if page_no > 1:
        params["paramJson"] = _json.dumps({"pageNo": page_no, "pageSize": 25}, ensure_ascii=False)
    try:
        resp = session.get(API_URL, params=params, timeout=30)
        data = resp.json()
        html = data["data"]["html"]
        # Extract count from pagination div
        count_m = re.search(r'count="(\d+)"', html)
        total = int(count_m.group(1)) if count_m else 0
        return html, total
    except Exception as e:
        print(f"  [API ERROR] page {page_no}: {e}")
        return None, 0


def parse_list(html):
    """解析API返回的列表HTML"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    content_div = soup.find(id="当前栏目列表")
    if not content_div:
        return items
    for li in content_div.select("li"):
        a = li.find("a")
        span = li.find("span")
        if not a or not a.get("href"):
            continue
        href = a.get("href", "").strip()
        title = a.get("title", "").strip() or a.get_text(strip=True)
        date_text = span.get_text(strip=True) if span else ""
        # Date format YYYY-MM-DD
        date_m = re.match(r"(\d{4}-\d{2}-\d{2})", date_text)
        date_str = date_m.group(1) if date_m else ""
        if href.startswith("/"):
            href = BASE_URL + href
        items.append((href, title, date_str))
    return items


def fetch_detail(url):
    try:
        resp = session.get(url, timeout=30)
        resp.encoding = "utf-8"
        return resp.text
    except Exception as e:
        print(f"ERR")
        return None


def extract_detail(html, url):
    soup = BeautifulSoup(html, "html.parser")
    # Title
    title = ""
    meta_title = soup.find("meta", attrs={"name": re.compile(r"ArticleTitle", re.I)})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            title = title_tag.get_text(strip=True).strip()

    # Date
    date_str = ""
    for name in ["PublishDate", "PubDate", "publishdate", "pubdate"]:
        meta_date = soup.find("meta", attrs={"name": re.compile(name, re.I)})
        if meta_date and meta_date.get("content"):
            raw = meta_date["content"].strip()
            m = re.match(r"(\d{4}-\d{2}-\d{2})", raw)
            if m:
                date_str = m.group(1)
                break

    # Content from div#zoom
    content_parts = []
    has_table = False
    zoom = soup.find("div", id="zoom", style=lambda v: v and "text-align" in v) or soup.find("div", id="zoom")
    if zoom:
        for child in zoom.children:
            if child.name == "p":
                text = child.get_text(strip=True)
                if text and text not in ("\xa0", ""):
                    content_parts.append(text)
            elif child.name == "table":
                has_table = True
                md = table_to_markdown(child)
                if md:
                    content_parts.append(md)
            elif child.name == "img":
                src = child.get("src", "")
                alt = child.get("alt", "")
                if src:
                    full_src = resolve_url(src)
                    content_parts.append(f"![{alt}]({full_src})")

    if not content_parts:
        # Fallback: all p in zoom
        if zoom:
            for p in zoom.find_all("p", recursive=True):
                text = p.get_text(strip=True)
                if text and text not in ("\xa0", ""):
                    content_parts.append(text)

    content = "\n\n".join(content_parts)

    # Attachments: direct files + API gateway download links
    attachments = []
    for a_tag in soup.find_all("a", href=re.compile(r"\.(doc|pdf|xls|docx|xlsx|rar|zip)$", re.I)):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True)
        if href:
            full_href = resolve_url(href)
            attachments.append({"name": text or os.path.basename(full_href), "url": full_href})
    # Also capture API gateway download links
    for a_tag in soup.find_all("a", href=re.compile(r"/api-gateway/jpaas-web-server/front/document/download", re.I)):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True)
        if href:
            full_href = resolve_url(href)
            attachments.append({"name": text or "附件", "url": full_href})
    attachments_json = _json.dumps(attachments, ensure_ascii=False) if attachments else ""

    return title, date_str, content, attachments_json, has_table


def resolve_url(url_str):
    if url_str.startswith("http"):
        return url_str
    if url_str.startswith("//"):
        return "https:" + url_str
    return BASE_URL + url_str


def table_to_markdown(table):
    rows = []
    for tr in table.find_all("tr"):
        cells = []
        for cell in tr.find_all(["td", "th"]):
            text = cell.get_text(strip=True)
            colspan = int(cell.get("colspan", 1))
            if colspan > 1:
                cells.extend([text] + [""] * (colspan - 1))
            else:
                cells.append(text)
        if cells:
            rows.append(cells)
    if not rows:
        return ""
    max_cols = max(len(r) for r in rows)
    for r in rows:
        while len(r) < max_cols:
            r.append("")
    lines = ["| " + " | ".join(rows[0]) + " |"]
    lines.append("| " + " | ".join(["---"] * max_cols) + " |")
    for r in rows[1:]:
        lines.append("| " + " | ".join(r) + " |")
    return "\n".join(lines)


def save_to_db(items_data):
    conn = sqlite3.connect(DB_PATH, timeout=10)
    c = conn.cursor()
    inserted = 0
    for title, date_str, content, attachments_json, has_table, page_url in items_data:
        try:
            c.execute("""
                INSERT OR REPLACE INTO gov_raw (page_url, title, site_name, publish_date, content, summary, attachments, has_table)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (
                page_url, title, SITE_NAME, date_str, content,
                content[:200] if content else "", attachments_json, 1 if has_table else 0
            ))
            inserted += 1
        except Exception as e:
            print(f"  [DB ERROR] {title[:30]}: {e}")
    conn.commit()
    conn.close()
    return inserted


def crawl():
    total_empty = 0
    total_segmented = 0
    total_attachments = 0
    empty_titles = []

    # 获取第一页，确认总记录数
    print(f"[{time.strftime('%H:%M:%S')}] 获取列表页...")
    html, total = fetch_list_page(1)
    if not html:
        print("列表页失败")
        return
    items_page1 = parse_list(html)
    total_pages = min((total + 24) // 25, MAX_PAGES)
    print(f"  总记录: {total}, 总页数: {total_pages}, 第一页: {len(items_page1)}条")

    # 获取所有页面
    all_items = list(items_page1)
    for page_no in range(2, total_pages + 1):
        html, _ = fetch_list_page(page_no)
        if html:
            items = parse_list(html)
            print(f"  第{page_no}/{total_pages}页: {len(items)}条")
            all_items.extend(items)
        else:
            print(f"  第{page_no}/{total_pages}页: 失败")
        time.sleep(0.5)

    print(f"\n共{len(all_items)}条待抓取")

    # 抓取详情
    batch_data = []
    for i, (detail_url, title, date_str) in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {title[:30]}...", end=" ", flush=True)
        html = fetch_detail(detail_url)
        if not html:
            print("F")
            continue
        detail_title, detail_date, content, attachments_json, has_table = extract_detail(html, detail_url)
        final_date = detail_date or date_str
        final_title = detail_title or title
        if not content:
            total_empty += 1
            empty_titles.append(final_title)
            print("E")
        elif "\n\n" in content:
            total_segmented += 1
            print("S")
        else:
            print("U")
        if attachments_json:
            try:
                total_attachments += len(_json.loads(attachments_json))
            except:
                pass
        batch_data.append((final_title, final_date, content, attachments_json, has_table, detail_url))
        time.sleep(0.5)

    # 入库
    if batch_data:
        saved = save_to_db(batch_data)
        print(f"\n入库: {saved}条")
    else:
        print("\n无数据入库")

    # 报告
    print(f"\n===== 报告 =====")
    print(f"总抓取: {len(batch_data)}")
    print(f"正文为空: {total_empty}")
    print(f"有分段: {total_segmented}/{len(batch_data)} ({total_segmented/max(len(batch_data),1)*100:.1f}%)")
    print(f"附件: {total_attachments}")
    if empty_titles:
        print(f"\n空正文标题:")
        for t in empty_titles[:10]:
            print(f"  - {t}")


if __name__ == "__main__":
    crawl()
