#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
垦利石化集团 - 通知公告爬虫
https://www.klsh.com/news/notice/
CMS: Discuz! X3.5 Portal
列表: ul.newlist > li (span a + p date)
分页: index.php?page=N (18页)
详情: div.new5 > div.c 正文
"""

import requests
import re
import sqlite3
import time
from bs4 import BeautifulSoup
BASE_URL = "https://www.klsh.com"
LIST_URL = "https://www.klsh.com/news/notice/"
SITE_NAME = "垦利石化-通知公告"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

DB_PATH = "/root/search.db"
MAX_PAGES = 5


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.text_factory = str
    return conn


def url_exists(conn, url):
    cur = conn.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (url,))
    return cur.fetchone() is not None


def insert_record(conn, record):
    conn.execute(
        """INSERT OR IGNORE INTO gov_raw
           (title, page_url, summary, content, site_name, publish_date)
           VALUES (?, ?, ?, ?, ?, ?)""",
        (record["title"], record["page_url"], record["summary"],
         record["content"], record["site_name"],
         record["publish_date"])
    )


def normalize_date(date_str):
    """标准化日期 YYYY-M-D -> YYYY-MM-DD"""
    if not date_str:
        return ""
    m = re.match(r'(\d{4})-(\d{1,2})-(\d{1,2})', date_str.strip())
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return date_str.strip()


def extract_clean_text(html):
    """提取格式良好的正文，保留段落结构"""
    # 剥离内联标签，防止字符被拆分（如"2026年6月9日"变成"202\n6\n年…"）
    text = re.sub(r'</?(b|span|font|strong|u|i|em|o:p)[^>]*>', '', html, flags=re.IGNORECASE)
    # 转换<br>为换行
    text = re.sub(r'<br\s*/?>', '\n', text)
    soup = BeautifulSoup(text, 'html.parser')
    return soup.get_text(separator='\n\n', strip=True)


def fetch_list(page):
    """获取列表页HTML"""
    if page == 1:
        url = LIST_URL
    else:
        url = f"{LIST_URL}index.php?page={page}"
    
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code == 200:
            return resp.text
        return None
    except Exception as e:
        print(f"  [ERROR] 列表页 {url} 请求失败: {e}")
        return None


def parse_list(html):
    """解析列表页，返回 (title, url, date) 列表"""
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.find("ul", class_="newlist")
    if not ul:
        return []
    
    items = []
    for li in ul.find_all("li"):
        a_tag = li.find("a")
        p_tag = li.find("p")
        if not a_tag:
            continue
        
        href = a_tag.get("href", "").strip()
        title = a_tag.get_text(strip=True).rstrip(" ...").rstrip("...")
        date_str = p_tag.get_text(strip=True) if p_tag else ""
        
        if not href or not title:
            continue
        
        # 补全URL
        if href.startswith("/"):
            full_url = BASE_URL + href
        elif href.startswith("http"):
            full_url = href
        else:
            full_url = BASE_URL + "/" + href
        
        # 提取日期（取日期部分，去掉时间）
        pub_date = normalize_date(date_str.split()[0] if date_str else "")
        
        items.append((title, full_url, pub_date))
    
    return items


def fetch_detail(url):
    """获取详情页正文"""
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code != 200:
            return None, []
        
        soup = BeautifulSoup(resp.text, "html.parser")
        
        # 正文在 div.new5 > div.c
        content_div = soup.find("div", class_="new5")
        if content_div:
            c_div = content_div.find("div", class_="c")
            if c_div:
                content_text = extract_clean_text(str(c_div))
            else:
                content_text = extract_clean_text(str(content_div))
        else:
            # Fallback
            c_div = soup.find("div", class_="c")
            if c_div:
                content_text = extract_clean_text(str(c_div))
            else:
                return None, []
        
        # 附件
        attachments = []
        parent_div = c_div if 'c_div' in dir() else content_div
        # Actually use the found div
        if content_div:
            for a in content_div.find_all("a"):
                href = a.get("href", "").strip()
                text = a.get_text(strip=True)
                if not href or href.startswith("#") or "javascript" in href.lower():
                    continue
                if href.startswith("/"):
                    ahref = BASE_URL + href
                elif not href.startswith("http"):
                    ahref = BASE_URL + "/" + href
                else:
                    ahref = href
                
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href.lower()):
                    attachments.append({"text": text, "url": ahref})
        
        return content_text, attachments
    
    except Exception as e:
        print(f"  [ERROR] 详情页 {url} 请求失败: {e}")
        return None, []


def main():
    print(f"=== 垦利石化-通知公告爬虫 ===\n目标: {LIST_URL}\n总页数: {MAX_PAGES}")

    conn = get_conn()

    total_found = 0
    total_inserted = 0
    total_skipped = 0

    for page in range(1, MAX_PAGES + 1):
        print(f"\n--- 第{page}页 ---")
        html = fetch_list(page)
        if not html:
            print(f"  [INFO] 第{page}页无内容，结束")
            break

        items = parse_list(html)
        if not items:
            print(f"  [INFO] 第{page}页无列表项，结束")
            break

        print(f"  本页 {len(items)} 条")

        for title, url, pub_date in items:
            total_found += 1

            if url_exists(conn, url):
                total_skipped += 1
                continue

            content, attachments = fetch_detail(url)
            if not content or len(content) < 50:
                short_title = title[:50]
                print(f"  [!] {short_title}... 内容过短")
                total_skipped += 1
                continue

            summary = content[:200] if len(content) > 200 else content

            record = {
                "title": title,
                "page_url": url,
                "summary": summary,
                "content": content,
                "site_name": SITE_NAME,
                "publish_date": pub_date,
            }

            insert_record(conn, record)
            conn.commit()
            total_inserted += 1

            short_title = title[:50]
            print(f"  [+] {short_title}... ({pub_date})")
            if attachments:
                print(f"     附件: {len(attachments)}个")

            time.sleep(0.5)

        time.sleep(1)

    conn.close()

    print(f"\n=== 完成 ===")
    print(f"总扫描: {total_found} 条")
    print(f"新增入库: {total_inserted} 条")
    print(f"已存在跳过: {total_skipped} 条")


if __name__ == "__main__":
    main()
