#!/usr/bin/env python3
"""
crawl_sanyue_tzgg.py - 山东三岳化工有限公司 公司新闻
CMS: 自定义PHP
列表: /index.php/companynews.html (ul.thumblist2 > li, 无分页)
详情: 无独立详情页，所有内容在列表页的 <p> 中
"""
import requests
import re
import sys
import os
import time

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
import hashlib
from crawler_lib import push_to_searchdb


def make_md5(text):
    return hashlib.md5(text.encode('utf-8')).hexdigest()

SITE_NAME = "山东三岳化工公司新闻"
BASE_URL = "https://www.sdsanyue.com"
LIST_URL = f"{BASE_URL}/index.php/companynews.html"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
}


def extract_articles(html):
    """从列表页提取所有文章（仅 ul.thumblist2 内）"""
    articles = []
    
    # 定位 ul.thumblist2 区域
    start = html.find('<ul class="thumblist2">')
    if start < 0:
        print("ERROR: 未找到 ul.thumblist2")
        return articles
    
    end = html.find('</ul>', start)
    if end < 0:
        print("ERROR: 未找到 ul.thumblist2 结束")
        return articles
    
    section = html[start:end + 5]
    
    # 匹配每个 li 项
    items = re.findall(
        r'<li[^>]*class="[^"]*(?:n1|n2)[^"]*"[^>]*>\s*(.*?)</li>',
        section, re.DOTALL
    )
    
    print(f"  发现 {len(items)} 个列表项")
    
    for item in items:
        # URL：只取 companynews 的
        url_match = re.search(r'href="(/index.php/companynews/\d+\.html)"', item)
        if not url_match:
            continue
        url = BASE_URL + url_match.group(1)
        
        # 标题：优先取 a.InfoPicture[title]（完整，无省略号）
        title = ""
        title_match = re.search(
            r'<a[^>]*class="InfoPicture"[^>]*title="([^"]+)"',
            item
        )
        if title_match:
            title = title_match.group(1).strip()
        else:
            # 回退取 span.InfoTitle
            title_match = re.search(
                r'<span class="InfoTitle">([^<]+)</span>',
                item
            )
            if title_match:
                title = title_match.group(1).strip()
        
        if not title:
            continue
        
        # 日期
        day = ""
        month = ""
        day_match = re.search(r'<span class="day">(\d+)</span>', item)
        if day_match:
            day = day_match.group(1)
        month_match = re.search(r'<span class="month">([^<]+)</span>', item)
        if month_match:
            month = month_match.group(1)
        
        pub_date = ""
        if month and day:
            pub_date = f"{month}-{day.zfill(2)}"
        
        # 正文：取 <p> 标签内容
        p_match = re.search(r'<p>(.*?)</p>', item, re.DOTALL)
        content_text = ""
        if p_match:
            raw = p_match.group(1).strip()
            content_text = raw
            content_text = re.sub(r'<br\s*/?>', '\n', content_text)
            content_text = re.sub(r'<[^>]+>', '', content_text)
            content_text = content_text.replace('&nbsp;', ' ').replace('&amp;', '&').replace('&lt;', '<').replace('&gt;', '>')
            content_text = re.sub(r'\s*\n\s*', '\n', content_text).strip()
        
        articles.append({
            "title": title,
            "url": url,
            "pub_date": pub_date,
            "content": content_text
        })
    
    return articles


def crawl(pages=1):
    """爬取公司新闻"""
    print(f"=== 山东三岳化工公司新闻 ===")
    print(f"列表页: {LIST_URL}")
    
    try:
        resp = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            print(f"ERROR: HTTP {resp.status_code}")
            return []
    except Exception as e:
        print(f"ERROR: {e}")
        return []
    
    articles = extract_articles(resp.text)
    print(f"  提取到 {len(articles)} 篇文章")
    
    results = []
    for art in articles:
        # 构建正文（纯文本 + 表格/附件标记）
        body = art["content"]
        
        # 附件检查（PDF/DOC/XLS链接在正文中已有保留）
        
        results.append({
            "title": art["title"],
            "url": art["url"],
            "content": body,
            "date": art["pub_date"],
            "site_name": SITE_NAME,
            "md5": make_md5(art["title"] + art["url"])
        })
    
    return results


def main():
    import argparse
    parser = argparse.ArgumentParser(description="山东三岳化工公司新闻爬虫")
    parser.add_argument("--pages", type=int, default=1, help="页数（本站无分页，固定全部）")
    parser.add_argument("--force", action="store_true", help="强制重复爬取")
    args = parser.parse_args()
    
    data = crawl(args.pages)
    if not data:
        print("未获取到数据")
        return
    
    print(f"\n入库 {len(data)} 条...")
    push_to_searchdb(data, SITE_NAME)
    print(f"完成！")


if __name__ == "__main__":
    main()
