#!/usr/bin/env python3
import os
"""
Sinopec SharePoint 站点通用爬虫
支持所有使用 Sinopec 标准 SharePoint 模板的分公司网站。

用法:
  python3 crawl_sinopec.py apw          # 安庆石化
  python3 crawl_sinopec.py bypc         # 燕山石化
  python3 crawl_sinopec.py zrcc         # 镇海石化
  python3 crawl_sinopec.py zrpc         # 中科炼化-公司新闻
  python3 crawl_sinopec.py zrpc_env     # 中科炼化-项目环保三同时
  python3 crawl_sinopec.py sjlh         # 石家庄炼化
  python3 crawl_sinopec.py ycfc         # 仪征化纤-环境保护
  python3 crawl_sinopec.py zhsh         # 中韩石化-公司公告
  python3 crawl_sinopec.py svw          # 川维化工-公司新闻
  python3 crawl_sinopec.py qlsh         # 齐鲁石化-公司公告
  python3 crawl_sinopec.py mmsh         # 茂名石化-公告公示
  python3 crawl_sinopec.py lysh         # 洛阳石化-安全环保
  python3 crawl_sinopec.py jnlh         # 济南炼化-信息公开
  python3 crawl_sinopec.py --all        # 全部站点
  python3 crawl_sinopec.py --incremental # 全部站点仅第1页(每日增量)

CMS: SharePoint，无WAF，可直接用 requests
"""

import sys, os, re, subprocess, time

# ========================================================================
# 站点配置
# ========================================================================
SITES = {
    "apw": {
        "name": "安庆石化",
        "domain": "apw.sinopec.com",
        "path": "/apw/csr/environmental_health",
        "pagination": "reverse",  # page N → default_{8-N}.shtml (默认 maxPage=7)
        "first_page_items": 28,
    },
    "bypc": {
        "name": "燕山石化",
        "domain": "bypc.sinopec.com",
        "path": "/bypc/csr/public_infor",
        "pagination": None,  # 单页
    },
    "zrcc": {
        "name": "镇海石化",
        "domain": "zrcc.sinopec.com",
        "path": "/zrcc/csr/safe_envir/lft_zxjk/eia_site",
        "pagination": "normal",  # page N → default_{N-1}.shtml
    },
    "zrpc": {
        "name": "中科炼化-公司新闻",
        "domain": "zrpc.sinopec.com",
        "path": "/zrpc/news/company_news",
        "pagination": None,
    },
    "zrpc_env": {
        "name": "中科炼化-项目环保三同时",
        "domain": "zrpc.sinopec.com",
        "path": "/zrpc/social_responsibility/environmental_protection/simulta_environ",
        "pagination": None,
    },
    "sjlh": {
        "name": "石家庄炼化",
        "domain": "sjlh.sinopec.com",
        "path": "/sjlh/csr/public_infor",
        "pagination": None,  # 单页，全部展示
    },
    # ========== 新增 7 站 (2026-05-18) ==========
    "ycfc": {
        "name": "仪征化纤-环境保护",
        "domain": "ycfc.sinopec.com",
        "path": "/ycfc/csr/envir_pro",
        "pagination": "reverse:3",  # maxPage=3, page N → default_{4-N}.shtml
    },
    "zhsh": {
        "name": "中韩石化-公司公告",
        "domain": "zhsh.sinopec.com",
        "path": "/zhsh/information/company_announce",
        "pagination": None,  # 单页
    },
    "svw": {
        "name": "川维化工-公司新闻",
        "domain": "svw.sinopec.com",
        "path": "/svw/news/com_news",
        "pagination": "reverse:16",  # maxPage=16
    },
    "qlsh": {
        "name": "齐鲁石化-公司公告",
        "domain": "qlsh.sinopec.com",
        "path": "/qlsh/news/com_notice",
        "pagination": "reverse:5",  # maxPage=5
    },
    "mmsh": {
        "name": "茂名石化-公告公示",
        "domain": "mmsh.sinopec.com",
        "path": "/mmsh/news/gggs",
        "pagination": "reverse:6",  # maxPage=6
    },
    "lysh": {
        "name": "洛阳石化-安全环保",
        "domain": "lysh.sinopec.com",
        "path": "/lysh/csr/safe_envir",
        "pagination": "reverse:2",  # maxPage=2
    },
    "jnlh": {
        "name": "济南炼化-信息公开",
        "domain": "jnlh.sinopec.com",
        "path": "/jnlh/xxgk",
        "pagination": "reverse:2",  # maxPage=2
    },
}

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
}
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")


def get_list_items(html, site_cfg):
    """从列表页HTML提取所有条目"""
    domain = site_cfg['domain']
    path = site_cfg['path']
    items = []
    
    # 方式1: title 属性存标题（大多数页面）
    pat1 = rf'<a[^>]*title="([^"]*)"[^>]*href="{re.escape(path)}/(\d+/news_[\w_]+\.shtml)"'
    for m in re.finditer(pat1, html):
        href = f"http://{domain}{path}/{m.group(2)}"
        title = m.group(1).strip()
        items.append({'title': title, 'href': href})
    
    # 方式2: a 标签文字做标题（部分分页页面）
    if not items:
        pat2 = rf'href="{re.escape(path)}/(\d+/news_[\w_]+\.shtml)"[^>]*>([^<]+)</a>'
        for m in re.finditer(pat2, html):
            href = f"http://{domain}{path}/{m.group(1)}"
            title = m.group(2).strip()
            if title:
                items.append({'title': title, 'href': href})
    
    return items


def get_detail(url, site_cfg, default_title=""):
    """访问详情页提取标题、日期、正文"""
    import requests
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        return {'error': str(e)}
    
    # 标题
    title = ""
    m = re.search(r'<div class="lfnews-title">(.*?)</div>', html, re.DOTALL)
    if m:
        title = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    if not title:
        m = re.search(r'<title>([^<]+)</title>', html)
        if m:
            title = re.sub(r'&nbsp;.*', '', m.group(1)).strip()
    if not title:
        title = default_title
    
    # 日期（从 URL 路径提取）
    pub_date = ""
    m = re.search(r'/(\d{8})/news_[\w_]+\.shtml', url)
    if m:
        d = m.group(1)
        pub_date = f"{d[:4]}-{d[4:6]}-{d[6:8]} 08:00:00"
    
    # 正文
    m = re.search(r'<div class="lfnews-content">(.*?)</div>\s*<div class="lfnews-bottom"', html, re.DOTALL)
    if m:
        content_html = m.group(1)
        # 去掉 SharePoint 控件标记
        content_html = re.sub(r'<div[^>]*style=\'display:none\'>[^<]*</div>', '', content_html)
        content_html = re.sub(r'<div[^>]*__ControlWrapper[^>]*>|</div>', '', content_html)
        content_html = re.sub(r'^\s*页面内容\s*', '', content_html)
        
        # 附件链接补全为绝对路径
        domain = site_cfg['domain']
        def make_abs_link(m):
            href = m.group(1)
            text = m.group(3)
            if href.startswith('/'):
                href = f'http://{domain}' + href
            return f'<a href="{href}">{text}</a>'
        content_html = re.sub(r'<a[^>]*href="([^"]+\.(pdf|docx?|xlsx?))"[^>]*>([^<]+)</a>',
                              make_abs_link, content_html, flags=re.I)
        
        # 清理内联样式和多余的 span
        content_html = re.sub(r' style="[^"]*"', '', content_html)
        content_html = re.sub(r'<span[^>]*>|</span>', '', content_html)
        content_html = re.sub(r'&nbsp;', ' ', content_html)
        
        content = content_html.strip()
    else:
        # 有些页面只包含 PDF 重定向链接（无标题和正文）
        m = re.search(r'url=(/[^"]+\.(pdf|docx?|xlsx?))', html)
        if m:
            doc_url = m.group(1)
            if doc_url.startswith('/'):
                doc_url = f'http://{site_cfg["domain"]}' + doc_url
            # 标题从列表页带入，这里用传入的 url 最终 title 会保留
            content = f'<p>附件文档：<a href="{doc_url}">点击查看/下载</a></p>'
        else:
            content = ""
    
    if not content or len(content) < 20:
        content = f'<p>{title}</p>'
    
    return {'title': title, 'pub_date': pub_date, 'content': content, 'url': url}


def get_page_url(site_cfg, page_n):
    """根据站点分页配置生成第N页URL"""
    domain = site_cfg['domain']
    path = site_cfg['path']
    base = f"http://{domain}{path}"
    
    if page_n == 1:
        return base + "/"
    
    pagination = site_cfg.get('pagination')
    
    if pagination and pagination.startswith('reverse'):
        # 格式: "reverse" (默认 maxPage=7) 或 "reverse:N"
        parts = pagination.split(':')
        max_page = int(parts[1]) if len(parts) > 1 else 7
        page_num = max_page + 1 - page_n
        if page_num < 1:
            return None
        return f"{base}/default_{page_num}.shtml"
    elif pagination == 'normal':
        # zrcc 格式: page N → default_{N-1}.shtml
        return f"{base}/default_{page_n-1}.shtml"
    else:
        # 单页或无分页
        return None



def crawl_site(site_key, incremental=False):
    """爬取指定站点的所有数据"""
    import requests

    cfg = SITES[site_key]
    site_name = cfg['name']

    print(f"\n爬取: {site_name}" + (" [增量模式]" if incremental else ""))

    all_details = []
    visited_urls = set()

    max_pages = 1 if incremental else 20
    for page_n in range(1, max_pages + 1):
        url = get_page_url(cfg, page_n)
        if not url:
            break
        
        print(f"  第{page_n}页: ", end="", flush=True)
        
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            html = r.text
        except Exception as e:
            print(f"请求失败: {e}")
            break
        
        items = get_list_items(html, cfg)
        if not items:
            print("0 条")
            break
        
        new_items = [i for i in items if i['href'] not in visited_urls]
        if not new_items or len(new_items) < len(items) * 0.3:
            print(f"{len(items)} 条 (已爬)")
            break
        
        print(f"{len(new_items)} 条")
        
        for item in new_items:
            title_short = item['title'][:50]
            print(f"    详情: {title_short}... ", end="", flush=True)
            
            detail = get_detail(item['href'], cfg, default_title=item['title'])
            visited_urls.add(item['href'])
            
            if detail.get('error'):
                print(f"FAIL: {detail['error']}")
                continue
            
            if len(detail.get('content', '')) < 20:
                print("内容过短，跳过")
                continue
            
            all_details.append(detail)
            print(f"OK ({len(detail['content'])}字)")
    
    if not all_details:
        print("未爬取到有效数据")
        return
    
    print(f"\n共获取 {len(all_details)} 条有效记录")
    
    # 写入服务器 search.db
    import sqlite3
    now = __import__('datetime').datetime.now().strftime('%Y-%m-%d %H:%M:%S')
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    
    new_count = 0
    for r in all_details:
        try:
            cur = conn.execute(
                'INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, summary, date_rank, status) VALUES (?, ?, ?, ?, ?, ?, ?, ?)',
                (site_name, r['url'], r['title'], r['content'], r['pub_date'], site_name, r['pub_date'], 'active')
            )
            if cur.rowcount > 0:
                new_count += 1
        except:
            pass
    
    conn.commit()
    total = conn.execute('SELECT COUNT(*) FROM gov_raw WHERE site_name=?', (site_name,)).fetchone()[0]
    conn.close()
    print(f"入库: 新增{new_count}, 累计{total}")
    
    print("完成")


def list_sites():
    """列出所有可用的站点"""
    print("可用站点:")
    for key, cfg in sorted(SITES.items()):
        pagination = cfg.get('pagination', '单页')
        print(f"  {key:12s} - {cfg['name']:20s} ({pagination})")


def main():
    if len(sys.argv) < 2:
        print("用法: python3 crawl_sinopec.py <站点key|--all> [--incremental]")
        print("示例:")
        print("  python3 crawl_sinopec.py apw          # 安庆石化")
        print("  python3 crawl_sinopec.py --all        # 全部站点")
        print("  python3 crawl_sinopec.py --incremental # 全部站点，仅第1页(每日增量)")
        list_sites()
        return

    incremental = '--incremental' in sys.argv
    site_key = sys.argv[1]

    if site_key in ('--all', '--incremental'):
        # 爬取所有站点
        for key in sorted(SITES.keys()):
            crawl_site(key, incremental)
        return

    if site_key == 'list':
        list_sites()
        return

    if site_key not in SITES:
        print(f"未知站点: {site_key}")
        list_sites()
        return

    crawl_site(site_key, incremental)


if __name__ == "__main__":
    main()
