#!/usr/bin/env python3
"""
天津环科源环保科技有限公司 (hky-ep.com) 爬虫
信息公告栏目 — direct_server 模式
"""
import re, sys, sqlite3, time, requests
import os

SITE_NAME = '天津环科源-信息公告'
BASE_URL = 'http://hky-ep.com'
LIST_URL = 'http://hky-ep.com/article_category.php?id=62'
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
    'Accept-Language': 'zh-CN,zh;q=0.9',
}

def fetch_list(page=0):
    url = f'{LIST_URL}&page={page}&s=zd=0'
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f'  [ERROR] 第{page+1}页: {e}')
        return []
    items, seen = [], set()
    for m in re.finditer(r'href="([^"]*article\.php\?id=(\d+))"', html):
        full_url = m.group(1)
        aid = m.group(2)
        if aid in seen: continue
        seen.add(aid)
        if not full_url.startswith('http'):
            full_url = f'{BASE_URL}/{full_url.lstrip("/")}'
        items.append({'title': '', 'url': full_url, 'aid': int(aid)})
    print(f'  第{page+1}页: 找到 {len(items)} 条')
    return items

def fetch_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except: return None
    result = {}
    mt = re.search(r'<title>([^<]+?)(?:\s*\|\s*信息公告\s*\|\s*文章中心\s*\|\s*天津环科源环保科技有限公司)?</title>', html)
    if mt: result['title'] = mt.group(1).strip()
    md = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if md: result['publish_date'] = md.group(1)
    mc = re.search(r'class="biaoti"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not mc: mc = re.search(r'class="left"[^>]*>(.*?)</div>', html, re.DOTALL)
    if mc:
        c = mc.group(1)
        c = re.sub(r'<span[^>]*>.*?</span>', '', c)
        c = re.sub(r'<div[^>]*class="time"[^>]*>.*?</div>', '', c, flags=re.DOTALL)
        c = c.strip()
        c = re.sub(r'href="(?!https?://)', f'href="{BASE_URL}/', c)
        c = re.sub(r'\s*style="[^"]*"', '', c)
        c = re.sub(r'\s*class="[^"]*"', '', c)
        result['content'] = c
    else: result['content'] = ''
    return result

def get_total_pages(html):
    pages = {int(m.group(1)) for m in re.finditer(r'page=(\d+)', html)}
    return max(pages) + 1 if pages else 1

def run(max_pages=None):
    print(f'{"="*50}')
    print(f'  {SITE_NAME}')
    print(f'{"="*50}')

    db = sqlite3.connect(SEARCH_DB, timeout=60)
    known = set(r[0] for r in db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall())
    db.close()

    try:
        resp = requests.get(LIST_URL, headers=HEADERS, timeout=30)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        print(f'  [ERROR] 首页: {e}')
        return

    total_pages = get_total_pages(html)
    print(f'  共 {total_pages} 页')
    if max_pages and max_pages < total_pages:
        total_pages = max_pages

    all_items = []
    for page in range(total_pages):
        items = fetch_list(page)
        for item in items:
            if item['url'] not in known:
                all_items.append(item)

    if not all_items:
        print("  no new data")
        return

    print(f'  抓取 {len(all_items)} 条详情...')
    results = []
    for i, item in enumerate(all_items):
        detail = fetch_detail(item['url'])
        title = detail.get('title', f'文章#{item["aid"]}') if detail else item['title']
        date_str = detail.get('publish_date', '') if detail else ''
        content = detail.get('content', '') if detail else ''
        summary = re.sub(r'<[^>]+>', ' ', content or '').strip()[:500]
        summary = re.sub(r'\s+', ' ', summary)
        results.append({
            "site_name": SITE_NAME, "source_url": item['url'][:500], "page_url": item['url'],
            "title": title[:500], "publish_date": date_str[:10] if date_str else "",
            "summary": title[:500], "content": content, "status": "active", "category": "", "tags": "",
        })
        time.sleep(0.3)

    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, skip = 0, 0
    for item in results:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name,source_url,page_url,title,publish_date,summary,content,status,category,tags) VALUES (?,?,?,?,?,?,?,?,?,?)",
                (item["site_name"][:200], item["source_url"], item["page_url"], item["title"],
                 item["publish_date"], item["summary"], item["content"], item["status"],
                 item["category"], item["tags"]),
            )
            if db.total_changes > 0: ok += 1
            else: skip += 1
        except: skip += 1
    db.commit()
    db.execute(
        "INSERT INTO gov_search(rowid,title,site_name,summary) SELECT r.id,r.title,r.site_name,r.summary FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,),
    )
    db.commit()
    db.close()
    print(f'  Saved: new={ok}, skip={skip}, FTS synced')

if __name__ == '__main__':
    mp = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(mp)
