#!/usr/bin/env python3
"""
招远市政府 - 建设项目环评审批 (zhaoyuan.gov.cn)
CMS: TRS/JCMS, AJAX 分页 (client-side only, 15 items)
API: /api-gateway/jpaas-publish-server/front/page/build/unit
  webId=155, pageId=48655, tagId=右侧文章列表
List: ul.bt-main-r-ul > li.bt-main-r-ul-li > a[title] + span(date)
Detail: /col/col48655/art/2026/art_{hash}.html
  Content: div.text#zoom (> p + table)
  Title: div.title > h1
  Date: span.time "发布日期：YYYY-MM-DD"
  Attachments: <a href="/api-gateway/...download-m?..."> inside content
  15 items/page, only page 1 available (AJAX client-side pagination)
"""
import sys
import re
import os
import json
import sqlite3
import urllib.request
import ssl

# === CONFIG ===
SITE_NAME = "招远市-建设项目环评审批"
BASE_URL = "https://www.zhaoyuan.gov.cn"
API_URL = (f"{BASE_URL}/api-gateway/jpaas-publish-server/front/page/build/unit"
           f"?parseType=bulidstatic&webId=155"
           f"&tplSetId=sW4TqQJo1K65B01yzYBRg&pageType=column"
           f"&tagId=%E5%8F%B3%E4%BE%A7%E6%96%87%E7%AB%A0%E5%88%97%E8%A1%A8"
           f"&pageId=48655&pageNo=1")
DB_PATH = "/root/search.db"
CUTOFF_DATE = "2023-01-01"
PAGES_DEFAULT = 1  # only 1 AJAX page available

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE


def fetch(url, referer=None):
    req = urllib.request.Request(url, headers={
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    })
    if referer:
        req.add_header("Referer", referer)
    try:
        resp = urllib.request.urlopen(req, timeout=20, context=ssl_ctx)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        print(f"  [FETCH ERROR] {url}: {e}")
        return None


def fetch_list_api():
    """Fetch list via AJAX API"""
    html = fetch(API_URL, referer="https://www.zhaoyuan.gov.cn/col/col48655/index.html")
    if not html:
        return []
    try:
        data = json.loads(html)
        d = data.get("data", {})
        if isinstance(d, dict):
            list_html = d.get("html", "")
        else:
            list_html = str(d)
    except json.JSONDecodeError:
        return []

    items = []
    pattern = r'<li[^>]*class="[^"]*bt-main-r-ul-li[^"]*">.*?<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>.*?<span[^>]*>(\d{4}-\d{2}-\d{2})</span>'
    for match in re.finditer(pattern, list_html, re.DOTALL):
        href = match.group(1).strip()
        title = match.group(2).strip()
        date = match.group(3).strip()
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append({"url": href, "title": title, "date": date})
    return items


def parse_detail(html):
    """Extract title, date, content from detail page"""
    title = ""
    date = ""
    content = ""

    # Title
    h1_match = re.search(r'<h1[^>]*>(.*?)</h1>', html)
    if h1_match:
        title = re.sub(r'<[^>]+>', '', h1_match.group(1)).strip()

    # Date
    date_match = re.search(r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if date_match:
        date = date_match.group(1)

    # Content from div.text#zoom
    zoom_idx = html.find('id="zoom"')
    if zoom_idx < 0:
        zoom_idx = html.find('class="text"')
    if zoom_idx > 0:
        start = html.rfind("<div", 0, zoom_idx)
        if start < 0:
            start = zoom_idx
        end_markers = ['<div class="other"', '<div class="qrcode', '<div class="footer', '<div id="footer']
        end = len(html)
        for marker in end_markers:
            pos = html.find(marker, zoom_idx)
            if 0 < pos < end:
                end = pos
        content_html = html[start:end]

        # Remove scripts
        text = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL)
        text = re.sub(r'<style[^>]*>.*?</style>', '', text, flags=re.DOTALL)

        # <br> → newline, </p> → double newline
        text = re.sub(r'<br\s*/?>', '\n', text)
        text = re.sub(r'</p>', '\n\n', text)
        text = text.replace('\r', '')

        # Convert tables to markdown
        tables = []
        for table_match in re.finditer(r'<table[^>]*>.*?</table>', text, re.DOTALL):
            table_html = table_match.group(0)
            rows = re.findall(r'<tr[^>]*>(.*?)</tr>', table_html, re.DOTALL)
            table_lines = []
            for row in rows:
                cells = re.findall(r'<t[dh][^>]*>(.*?)</t[dh]>', row, re.DOTALL)
                cell_texts = [re.sub(r'<[^>]+>', '', c).strip() for c in cells]
                if cell_texts and any(cell_texts):
                    table_lines.append("| " + " | ".join(cell_texts) + " |")
            if table_lines:
                if len(table_lines) >= 2:
                    hdr = table_lines[0].count("|") - 1
                    table_lines.insert(1, "|" + "---|" * hdr)
                tables.append("\n".join(table_lines))

        # Get clean text
        clean = re.sub(r'<[^>]+>', '', text)
        clean = re.sub(r'[ \t]+', ' ', clean)
        clean = re.sub(r' *\n *', '\n', clean)
        clean = re.sub(r'\n{3,}', '\n\n', clean)
        clean = clean.strip()

        parts = [clean] if clean else []
        if tables:
            parts.append("\n\n".join(tables))
        content = "\n\n".join(parts)

    # Attachments (PDF via download API)
    attachments = []
    for match in re.finditer(r'<a[^>]*href="([^"]*download-m[^"]*)"[^>]*>([^<]*)</a>', html):
        href = match.group(1)
        text = match.group(2).strip()
        if not href.startswith("http"):
            href = BASE_URL + href
        attachments.append({"href": href, "text": text or href.split("=")[-1][:50]})

    return {"title": title, "date": date, "content": content, "attachments": attachments}


def insert_to_db(items):
    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    inserted = 0
    skipped = 0

    for item in items:
        title = item.get("title", "")
        url = item.get("url", "")
        date = item.get("date", "")
        content = item.get("content", "")

        attach_str = ""
        for att in item.get("attachments", []):
            if attach_str:
                attach_str += "\n"
            attach_str += f"[{att['text']}]({att['href']})"

        if not content and not title:
            skipped += 1
            continue

        try:
            c.execute("""
                INSERT OR IGNORE INTO gov_raw
                (title, page_url, site_name, publish_date, content, summary, attachments, source_url, date_rank)
                VALUES (?, ?, ?, ?, ?, '', ?, ?, CAST(strftime('%s', ?) AS INTEGER))
            """, (title, url, SITE_NAME, date, content, attach_str, url, date))
            if c.rowcount > 0:
                inserted += 1
            else:
                skipped += 1
        except Exception as e:
            print(f"  [DB ERROR] {title[:30]}: {e}")
            skipped += 1

    conn.commit()
    conn.close()
    return inserted, skipped


def main():
    max_pages = PAGES_DEFAULT
    for arg in sys.argv[1:]:
        if arg.isdigit():
            max_pages = int(arg)

    print(f"[INFO] {SITE_NAME} - 爬虫")

    all_items = []
    total_new = 0
    total_old = 0

    # Fetch list (only 1 AJAX page)
    print(f"\n  Fetching list via AJAX API...")
    items = fetch_list_api()
    print(f"  Found {len(items)} items")

    if not items:
        print(f"  [STOP] No items")
        return

    for item in items:
        if item["date"] < CUTOFF_DATE:
            print(f"  [STOP] Date {item['date']} < {CUTOFF_DATE}")
            break

        print(f"    {item['date']} {item['title'][:50]}...")
        detail_html = fetch(item["url"])
        if not detail_html:
            print(f"    [SKIP] Cannot fetch detail")
            total_old += 1
            continue

        detail = parse_detail(detail_html)
        if detail["title"]:
            item["title"] = detail["title"]
        if detail["date"]:
            item["date"] = detail["date"]
        item["content"] = detail["content"]
        item["attachments"] = detail.get("attachments", [])
        all_items.append(item)

    if all_items:
        new, old = insert_to_db(all_items)
        total_new += new
        total_old += old
        print(f"  [DB] +{new} new, {old} existing")

    print(f"\n[DONE] 新增: {total_new}, 跳过: {total_old}")

    if total_new > 0:
        conn = sqlite3.connect(DB_PATH, timeout=30)
        conn.execute("INSERT INTO gov_search(gov_search) VALUES('rebuild')")
        conn.commit()
        conn.close()
        print("FTS rebuilt")


if __name__ == "__main__":
    main()
