#!/usr/bin/env python3
"""
昆山环境保护产业协会 - 企业信息公开爬虫
URL: http://www.ksepia.com/page104?article_category=29
CMS: 建站之星 (ysjianzhan.cn)
列表: POST index.php?_m=article_list&_a=get_page (AJAX分页, 8条/页, 共118页)
详情: /page108?article_id=N
内容: PDF附件为主, 部分有HTML正文(含表格)
日跑: --max-pages 1 (增量取最新一页)
"""

import sys, os, re, time, sqlite3, argparse, json
from datetime import datetime
from urllib.parse import urljoin
import requests
from bs4 import BeautifulSoup

import warnings
warnings.filterwarnings("ignore")

BASE_URL = "http://www.ksepia.com"
SITE_NAME = os.environ.get("SITE_NAME", "昆山环境保护产业协会")
CATEGORY = os.environ.get("CATEGORY_LABEL", "企业信息公开")
GROUP = "企业"
DB_PATH = os.environ.get("DB_PATH", "/mnt/data/search.db")
CATEGORY_ID = os.environ.get("CATEGORY_ID", "29")
LAYER_ID = os.environ.get("LAYER_ID", "layerA4DA9CC88623AA5CAB87284C0AE5EF15")
PAGE_SIZE = 8

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
}


def get_session():
    """获取带 PHPSESSID 的 session"""
    s = requests.Session()
    s.headers.update(HEADERS)
    # 先访问列表页获取 cookie
    s.get(f"{BASE_URL}/page104?article_category={CATEGORY_ID}", timeout=30)
    return s


def fetch_list(session, page):
    """通过 AJAX API 获取文章列表"""
    url = f"{BASE_URL}/index.php?_m=article_list&_a=get_page"
    data = {
        "article_category": CATEGORY_ID,
        "layer_id": LAYER_ID,
        "page": page,
        "article_category_more": CATEGORY_ID,
    }
    items = []
    try:
        resp = session.post(url, data=data, timeout=30)
        resp.raise_for_status()
        html = resp.text

        # 解析 li 列表
        soup = BeautifulSoup(html, "html.parser")
        for li in soup.select("li.wpart-border-line"):
            a_tag = li.select_one("a.articleid")
            if not a_tag:
                continue
            href = a_tag.get("href", "")
            title = (a_tag.get("title") or a_tag.get_text(strip=True) or "").strip()
            if not title or not href:
                continue

            # 提取 article_id
            m = re.search(r'article_id=(\d+)', href)
            if not m:
                continue
            article_id = m.group(1)
            detail_url = f"{BASE_URL}/page108?article_id={article_id}"

            items.append({
                "url": detail_url,
                "title": title,
                "article_id": article_id,
            })
        return items
    except Exception as e:
        print(f"  [错误] 列表页 {page} 请求失败: {e}", flush=True)
        return []


def fetch_detail(session, detail_url, article_id):
    """获取详情页标题、日期和正文"""
    try:
        resp = session.get(detail_url, timeout=30)
        resp.encoding = "utf-8"
        html = resp.text

        # 标题
        title = ""
        m = re.search(r'artdetail_title[^>]*>\s*([^<]+)', html)
        if m:
            title = m.group(1).strip()

        # 日期
        publish_date = ""
        m = re.search(r'发布时间:\s*</span>\s*([\d-]+)', html)
        if m:
            publish_date = m.group(1).strip()

        # 正文 - 定位 setsid="articleXXXX" 的 div (跳过 CSS 中的 artview_detail 字符串)
        body = ""
        idx = html.find('setsid="article')
        if idx >= 0:
            # 找到最近的 <div 前，确定这是 artview_detail div
            start = html.rfind('<div', 0, idx)
            if start < 0:
                start = idx
            # 从这个 <div 开始查找 </div> 结束 (注意嵌套)
            depth = 1
            i = html.find('>', start) + 1
            while i < len(html) and depth > 0:
                if html[i:i+4] == '<div' and not html[i+4].isspace() and html[i+4] not in '>':
                    # 是嵌套 div（如 <div style="clear">）
                    depth += 1
                    i += 4
                elif html[i:i+6] == '</div>':
                    depth -= 1
                    i += 6
                else:
                    i += 1
            raw_content = html[start:i]
            # 提取 <div class="artview_detail"> 内的 HTML
            inner_start = raw_content.find('>', raw_content.find('artview_detail')) + 1
            inner_end = raw_content.rfind('</div>')
            if inner_start > 0 and inner_end > inner_start:
                body = raw_content[inner_start:inner_end].strip()
            else:
                body = raw_content

        # 检查是否含链接 (PDF 附件)
        if not body or len(body) < 20:
            return title, publish_date, ""

        return title, publish_date, body

    except Exception as e:
        print(f"  [错误] 详情页 {detail_url}: {e}", flush=True)
        return "", "", ""


def clean_content_html(html_content):
    """清洗正文 HTML"""
    if not html_content:
        return ""

    soup = BeautifulSoup(html_content, "html.parser")

    # 处理图片路径
    for img in soup.find_all("img"):
        src = img.get("src", "")
        if src and not src.startswith("http"):
            img["src"] = urljoin(BASE_URL, src)

    # 处理附件链接
    for a in soup.find_all("a"):
        href = a.get("href", "")
        if href and not href.startswith("http") and not href.startswith("#") and not href.startswith("javascript"):
            a["href"] = urljoin(BASE_URL, href)
        # 去掉空链接和 # 链接
        if href in ("#", ""):
            a.decompose()

    result = str(soup)
    # 清理多余样式
    result = re.sub(r'\sstyle="[^"]*"', '', result)
    result = re.sub(r'<span[^>]*>|</span>', '', result)
    result = re.sub(r'<strong[^>]*>|</strong>', '', result)
    result = re.sub(r'<b[^>]*>|</b>', '', result)
    result = re.sub(r'<font[^>]*>|</font>', '', result)
    result = re.sub(r'\n{3,}', '\n\n', result)
    return result.strip()


def crawl(max_pages=None):
    s = get_session()

    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")

    existing = set()
    cur = db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    for row in cur:
        existing.add(row[0])

    print(f"已有 {len(existing)} 条 {SITE_NAME} 记录", flush=True)

    # 先取第一页确定总页数 - 从 AJAX 返回的 pager 信息中获取
    first_items = fetch_list(s, 1)
    if not first_items:
        print("无法获取列表数据", flush=True)
        db.close()
        return 0

    # 从页面 HTML 中获取总页数 (118)
    total_pages = 118
    pages_to_fetch = min(max_pages, total_pages) if max_pages else 5
    print(f"共约 {total_pages} 页, 爬取前 {pages_to_fetch} 页", flush=True)

    total_new = 0
    total_skip = 0
    total_error = 0

    for page in range(1, pages_to_fetch + 1):
        print(f"[第 {page}/{pages_to_fetch} 页] ", end="", flush=True)
        items = first_items if page == 1 else fetch_list(s, page)
        if not items:
            print("无数据", flush=True)
            continue

        page_new = 0
        for item in items:
            full_url = item["url"]
            if full_url in existing:
                total_skip += 1
                continue

            article_id = item["article_id"]
            title = item["title"]

            if not title:
                total_skip += 1
                continue

            # 获取详情
            detail_title, publish_date, body = fetch_detail(s, full_url, article_id)

            # 标题：详情优先，列表作为备选
            final_title = detail_title if detail_title else title

            # 清洗正文 HTML
            if body and len(body.strip()) > 20:
                body = clean_content_html(body)

            if not body or len(body.strip()) < 30:
                body = ""

            # 摘要
            summary = re.sub(r'<[^>]+>', ' ', body) if body else title
            summary = re.sub(r'\s+', ' ', summary).strip()[:300]

            has_table = 1 if body and '<table' in body else 0

            # 日期
            if not publish_date or len(publish_date) < 10:
                publish_date = ""

            try:
                cur = db.execute("""INSERT OR IGNORE INTO gov_raw
                    (site_name, source_url, page_url, title, publish_date,
                     summary, content, status, category, group_name, has_table)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""", (
                    SITE_NAME, full_url, full_url, final_title[:500], publish_date,
                    summary[:500], body if body else '',
                    'active', CATEGORY, GROUP, has_table
                ))
                if cur.rowcount > 0:
                    row = db.execute("SELECT id FROM gov_raw WHERE page_url=?", (full_url,)).fetchone()
                    if row:
                        rid = row[0]
                        try:
                            # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会因
                            #   rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
                            db.commit()
                            db.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                                       (rid, final_title[:500], SITE_NAME, summary[:200]))
                        except sqlite3.IntegrityError:
                            pass
                    existing.add(full_url)
                    db.commit()
                    total_new += 1
                    page_new += 1
                else:
                    total_skip += 1
            except Exception as e:
                print(f"x", end="", flush=True)
                total_error += 1

        print(f"+{page_new} 条 (累计 {total_new})", flush=True)
        if page < pages_to_fetch:
            time.sleep(0.5)

    db.close()
    print(f"\n完成! 新增 {total_new}, 跳过 {total_skip}, 错误 {total_error}", flush=True)
    return total_new


if __name__ == "__main__":
    parser = argparse.ArgumentParser()
    parser.add_argument("--max-pages", type=int, default=None)
    args = parser.parse_args()
    crawl(max_pages=args.max_pages)
