#!/usr/bin/env python3
"""
新疆蓝山屯河科技股份有限公司 (www.lanshantunhe.com) 爬虫
新闻中心 → 专题专栏 — direct_server 模式
"""
import re, sys, sqlite3, time, requests
import os

SITE_NAME = '蓝山屯河-专题专栏'
BASE_URL = 'https://www.lanshantunhe.com'
LIST_URL = 'https://www.lanshantunhe.com/news/special/'
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9',
}

def fetch_list(page=1):
    url = LIST_URL if page == 1 else f'{BASE_URL}/news/special/p{page}.html'
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except:
        return []
    seen, items = set(), []
    for m in re.finditer(r'href="(/news/special/(\d+)\.html)"', html):
        href = m.group(1)
        if href in seen: continue
        seen.add(href)
        title_m = re.search(rf'href="{re.escape(href)}"[^>]*>([^<]+)</a>', html)
        title = title_m.group(1).strip() if title_m else ''
        if not title: continue
        items.append({'title': title, 'url': f'{BASE_URL}{href}'})
    print(f"  第{page}页: {len(items)} 条")
    return items

def fetch_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except: return None
    result = {}
    title_m = re.search(r'<title>([^<]+?)(?:_专题专栏_新闻中心)?(?:_新疆蓝山屯河科技股份有限公司)?</title>', html)
    if title_m: result['title'] = title_m.group(1).strip()
    date_m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if date_m: result['publish_date'] = date_m.group(1)
    content_m = re.search(r'<div class="ctbox"[^>]*>(.*?)</div>\s*<div class="shangxia"', html, re.DOTALL)
    if content_m:
        c = content_m.group(1).strip()
        c = re.sub(r'src="(?!https?://)', f'src="{BASE_URL}', c)
        c = re.sub(r'href="(?!https?://)(?!/)', f'href="{BASE_URL}/', c)
        c = re.sub(r'\s*style="[^"]*"', '', c)
        c = re.sub(r'\s*class="[^"]*"', '', c)
        result['content'] = c
    else: result['content'] = ''
    return result

def get_total_pages(html):
    pages = [int(n) for n in re.findall(r'data-ci-pagination-page="(\d+)"', html)]
    return max(pages) if pages else 1

def run(max_pages=None):
    print(f'{"="*50}')
    print(f'  {SITE_NAME}')
    print(f'{"="*50}')

    db = sqlite3.connect(SEARCH_DB, timeout=60)
    known = set(r[0] for r in db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall())
    db.close()

    try:
        resp = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f"  [ERROR] 首页: {e}")
        return

    total = get_total_pages(html)
    print(f"  共 {total} 页")
    if max_pages and max_pages < total:
        total = max_pages

    all_items = []
    for page in range(1, total + 1):
        items = fetch_list(page)
        for item in items:
            if item['url'] not in known:
                all_items.append(item)

    if not all_items:
        print("  no new data")
        return

    print(f'  抓取 {len(all_items)} 条详情...')
    results = []
    for i, item in enumerate(all_items):
        detail = fetch_detail(item['url'])
        title = detail.get('title', '') if detail else item['title']
        if not title: title = item['title']
        date_str = detail.get('publish_date', '') if detail else ''
        content = detail.get('content', '') if detail else ''
        summary = re.sub(r'<[^>]+>', ' ', content or '').strip()[:500]
        summary = re.sub(r'\s+', ' ', summary)
        results.append({
            "site_name": SITE_NAME, "source_url": item['url'][:500], "page_url": item['url'],
            "title": title[:500], "publish_date": date_str[:10] if date_str else "",
            "summary": title[:500], "content": content, "status": "active", "category": "", "tags": "",
        })
        time.sleep(0.3)

    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, skip = 0, 0
    for item in results:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name,source_url,page_url,title,publish_date,summary,content,status,category,tags) VALUES (?,?,?,?,?,?,?,?,?,?)",
                (item["site_name"][:200], item["source_url"], item["page_url"], item["title"],
                 item["publish_date"], item["summary"], item["content"], item["status"],
                 item["category"], item["tags"]),
            )
            if db.total_changes > 0: ok += 1
            else: skip += 1
        except: skip += 1
    db.commit()
    # 2026-09-22: 先提交 gov_raw —— 库上触发器已维护 FTS，下面这条手动写入会
    #   因 rowid 重复而 IntegrityError；不先 commit 会把 gov_raw 那条一并回滚（静默丢数据）
    db.commit()
    db.execute(
        "INSERT OR REPLACE INTO gov_search(rowid,title,site_name,summary) SELECT r.id,r.title,r.site_name,r.summary FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,),
    )
    db.commit()
    db.close()
    print(f'  Saved: new={ok}, skip={skip}, FTS synced')

if __name__ == '__main__':
    mp = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(mp)
