#!/usr/bin/env python3
"""
舟山高新技术产业园区管理委员会 - 通知公告爬虫
URL: http://gxq.zhoushan.gov.cn/col/col1229339186/index.html
CMS: JCMS (集团网站群) - 浙江省政府网站群
列表: API GET /api-gateway/jpaas-publish-server/front/page/build/unit (10条/页, 共60条6页)
详情: /col/col1229339186/art/YYYY/art_XXXXXXXXXXXX.html (静态HTML, 含#zoom正文)
日跑: --max-pages 1 (增量取最新一页)
"""

import sys, os, re, time, sqlite3, argparse, json
from datetime import datetime
from urllib.parse import urljoin, urlencode
import requests
from bs4 import BeautifulSoup

import warnings
warnings.filterwarnings("ignore")

BASE_URL = "http://gxq.zhoushan.gov.cn"
SITE_NAME = "舟山高新技术产业园区管理委员会"
CATEGORY = "通知公告"
GROUP = "浙江省"
DB_PATH = os.environ.get("DB_PATH", "/mnt/data/search.db")
COLUMN_ID = "1229339186"
WEB_ID = "3062"
TPL_SET_ID = "BS3nV4yH2bkBfJ7b85psF"
PAGE_SIZE = 10

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{BASE_URL}/col/col{COLUMN_ID}/index.html",
}


def fetch_list(page):
    """通过 API 获取文章列表"""
    url = f"{BASE_URL}/api-gateway/jpaas-publish-server/front/page/build/unit"
    param_json = json.dumps({"pageNo": page, "pageSize": PAGE_SIZE}, ensure_ascii=False)
    params = {
        "parseType": "bulidstatic",
        "webId": WEB_ID,
        "tplSetId": TPL_SET_ID,
        "pageType": "column",
        "tagId": "文章列表或正文",
        "editType": "null",
        "pageId": COLUMN_ID,
        "paramJson": param_json,
    }
    items = []
    try:
        resp = requests.get(url, params=params, headers=HEADERS, timeout=30)
        resp.raise_for_status()
        data = resp.json()
        html = data["data"]["html"]

        soup = BeautifulSoup(html, "html.parser")
        for li in soup.select("li"):
            a_tag = li.select_one("a[href]")
            if not a_tag:
                continue
            href = a_tag.get("href", "")
            title = (a_tag.get("title") or a_tag.get_text(strip=True) or "").strip()
            if not title or not href or len(title) < 5:
                continue
            if href.startswith("/"):
                href = f"{BASE_URL}{href}"
            elif not href.startswith("http"):
                href = f"{BASE_URL}/{href.lstrip('/')}"

            # 列表日期: day=<day> date=<YYYY-MM>
            day_el = li.select_one(".day")
            date_el = li.select_one(".date")
            publish_date = ""
            if date_el and day_el:
                ym = date_el.get_text(strip=True)
                d = day_el.get_text(strip=True)
                if ym and d:
                    publish_date = f"{ym}-{d.zfill(2)}"
            if not publish_date:
                for span in li.select("span"):
                    txt = span.get_text(strip=True)
                    m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
                    if m:
                        publish_date = m.group(1)
                        break

            items.append({
                "url": href,
                "title": title,
                "date": publish_date,
            })
        return items
    except Exception as e:
        print(f"  [错误] 列表页 {page} 请求失败: {e}", flush=True)
        return []


def fetch_detail(detail_url):
    """获取详情页标题、日期和正文"""
    try:
        resp = requests.get(detail_url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text

        # 标题: <h3> inside .article_intro
        title = ""
        m = re.search(r'<h3[^>]*>\s*(.*?)\s*</h3>', html, re.DOTALL)
        if m:
            title = re.sub(r'<[^>]+>', '', m.group(1)).strip()

        # 日期: <span class="article_time">YYYY-MM-DD</span>
        publish_date = ""
        m = re.search(r'article_time[^>]*>\s*(\d{4}-\d{2}-\d{2})', html)
        if m:
            publish_date = m.group(1).strip()

        # 正文: <div class="ia-text" id="zoom">
        body = ""
        zoom_idx = html.find('id="zoom"')
        if zoom_idx >= 0:
            start = html.find('>', zoom_idx) + 1
            depth = 1
            i = start
            while i < len(html) and depth > 0:
                if html[i:i+4] == '<div':
                    depth += 1
                    i += 4
                elif html[i:i+6] == '</div>':
                    depth -= 1
                    i += 6
                else:
                    i += 1
            raw = html[start:i-6] if depth == 0 else html[start:start+5000]
            if raw.strip():
                body = clean_content_html(raw.strip())

        if not body or len(body) < 20:
            body = ""

        return title, publish_date, body

    except Exception as e:
        print(f"  [错误] 详情页 {detail_url}: {e}", flush=True)
        return "", "", ""


def clean_content_html(html_content):
    """清洗正文 HTML"""
    if not html_content:
        return ""

    soup = BeautifulSoup(html_content, "html.parser")

    # 处理图片路径
    for img in soup.find_all("img"):
        src = img.get("src", "")
        if src and not src.startswith("http"):
            img["src"] = urljoin(BASE_URL, src)

    # 处理附件链接
    for a in soup.find_all("a"):
        href = a.get("href", "")
        if href and not href.startswith("http") and not href.startswith("#") and not href.startswith("javascript"):
            a["href"] = urljoin(BASE_URL, href)

    result = str(soup)
    # 清理多余样式标签
    result = re.sub(r'\sstyle="[^"]*"', '', result)
    result = re.sub(r'<span[^>]*>|</span>', '', result)
    result = re.sub(r'<strong[^>]*>|</strong>', '', result)
    result = re.sub(r'<b[^>]*>|</b>', '', result)
    result = re.sub(r'<font[^>]*>|</font>', '', result)
    result = re.sub(r'\n{3,}', '\n\n', result)
    return result.strip()


def crawl(max_pages=None):
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    existing = set()
    cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    for row in cur:
        existing.add(row[0])

    print(f"已有 {len(existing)} 条 {SITE_NAME} 记录", flush=True)

    first_items = fetch_list(1)
    if not first_items:
        print("无法获取列表数据", flush=True)
        db.close()
        return 0

    # 总量 ~60 条, 6 页
    total_pages = 6
    pages_to_fetch = min(max_pages, total_pages) if max_pages else 5
    print(f"共约 {total_pages} 页, 爬取前 {pages_to_fetch} 页", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, pages_to_fetch + 1):
        print(f"[第 {page}/{pages_to_fetch} 页] ", end="", flush=True)
        items = first_items if page == 1 else fetch_list(page)
        if not items:
            print("无数据", flush=True)
            continue

        page_new = 0
        for item in items:
            full_url = item["url"]
            if full_url in existing:
                total_skip += 1
                continue

            title = item["title"]
            publish_date = item["date"]

            if not title:
                total_skip += 1
                continue

            # 获取详情（补充标题/日期/正文）
            det_title, det_date, body = fetch_detail(full_url)
            final_title = det_title if det_title else title
            if det_date and not publish_date:
                publish_date = det_date

            # 摘要
            summary = re.sub(r'<[^>]+>', ' ', body) if body else final_title
            summary = re.sub(r'\s+', ' ', summary).strip()[:300]

            has_table = 1 if body and '<table' in body else 0

            try:
                cur = db.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date,
                     summary, content, status, category, group_name, has_table)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                    SITE_NAME, full_url, full_url, final_title[:500], publish_date[:10],
                    summary[:500], body if body else '',
                    'active', CATEGORY, GROUP, has_table
                ))
                if cur.rowcount > 0:
                    row = db.execute("SELECT id FROM gov_raw WHERE page_url=?", (full_url,)).fetchone()
                    if row:
                        rid = row[0]
                        try:
                            db.execute("INSERT INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                                       (rid, final_title[:500], SITE_NAME, summary[:200]))
                        except sqlite3.IntegrityError:
                            pass
                    existing.add(full_url)
                    db.commit()
                    total_new += 1
                    page_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"x", end="", flush=True)
                total_error += 1

        print(f"+{page_new} 条 (累计 {total_new})", flush=True)
        if page < pages_to_fetch:
            time.sleep(0.5)

    db.close()
    print(f"\n完成! 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error}", flush=True)
    return total_new


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None)
    args = parser.parse_args()
    crawl(max_pages=args.max_pages)
