#!/usr/bin/env python3
"""
中煤平朔集团有限公司 - 公告公示爬虫
API分页 — direct_server 模式
"""
import requests, json, re, sqlite3, sys, time
import os

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "中煤平朔公告公示"
BASE = "https://pingshuo.chinacoal.com"
API_URL = BASE + "/api-gateway/jpaas-publish-server/front/page/build/unit"
MAX_PAGES = 5

def fetch_list(page_no, page_size=15):
    params = {
        "parseType": "bulidstatic", "webId": "18",
        "tplSetId": "jGM7GMyKBL1qIYcHxcFU9", "pageType": "column",
        "tagId": "当前栏目_list", "editType": "null", "pageId": "1508",
        "paramJson": json.dumps({"pageNo": page_no, "pageSize": page_size}, ensure_ascii=False)
    }
    try:
        r = requests.get(API_URL, params=params, timeout=30)
        r.raise_for_status()
        return r.json().get("data", {}).get("html", "")
    except: return ""

def parse_list(html):
    items = []
    for li in re.findall(r'<li>(.*?)</li>', html, re.DOTALL):
        title_m = re.search(r'title="([^"]*)"', li)
        href_m = re.search(r'href="([^"]*)"', li)
        date_m = re.search(r'(\d{4}-\d{2}-\d{2})', li)
        if title_m and href_m and date_m:
            items.append((title_m.group(1), BASE + href_m.group(1), date_m.group(1)))
    return items

def fetch_detail(page_url):
    try:
        r = requests.get(page_url, timeout=30)
        r.encoding = 'utf-8'
        html = r.text
    except: return "", "", ""

    title = ""
    m = re.search(r'<title>(.*?)</title>', html)
    if m: title = m.group(1).strip()

    publish_date = ""
    m = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    if m: publish_date = m.group(1)

    content = ""
    body = re.search(r'<!--begin-->([\s\S]*?)<!--end-->', html)
    if body:
        text = re.sub(r'<[^>]+>', '', body.group(1)).strip()
        text = re.sub(r'\s+', ' ', text).strip()
        if len(text) > 20: content = text
    if not content and title: content = title
    return title, publish_date, content

def run(max_pages=None):
    if max_pages is None: max_pages = MAX_PAGES
    print(f"  {SITE_NAME}")

    db = sqlite3.connect(SEARCH_DB, timeout=60)
    known = set(r[0] for r in db.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchall())
    db.close()

    all_items = []
    for page_no in range(1, max_pages + 1):
        print(f"  Page {page_no}...", end=" ", flush=True)
        html = fetch_list(page_no)
        if not html:
            print("FAIL")
            break
        items = parse_list(html)
        new_items = [(t, u, d) for t, u, d in items if u not in known]
        print(f"OK {len(items)} items (new: {len(new_items)})")
        all_items.extend(new_items)
        if len(items) < 10: break

    if not all_items:
        print("  no new data")
        return

    print(f"  Fetching {len(all_items)} details...")
    results = []
    for i, (title, url, date_str) in enumerate(all_items):
        dt, pd, content = fetch_detail(url)
        final_title = dt or title
        final_date = pd or date_str
        summary = re.sub(r'<[^>]+>', ' ', content or '').strip()[:500]
        results.append({
            "site_name": SITE_NAME, "source_url": url[:500], "page_url": url,
            "title": final_title[:500], "publish_date": final_date[:10] if final_date else "",
            "summary": final_title[:500], "content": content, "status": "active", "category": "", "tags": "",
        })
        if (i + 1) % 10 == 0: print(f"    [{i+1}/{len(all_items)}]")
        time.sleep(0.3)

    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=NORMAL")
    ok, skip = 0, 0
    for item in results:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name,source_url,page_url,title,publish_date,summary,content,status,category,tags) VALUES (?,?,?,?,?,?,?,?,?,?)",
                (item["site_name"][:200], item["source_url"], item["page_url"], item["title"],
                 item["publish_date"], item["summary"], item["content"], item["status"],
                 item["category"], item["tags"]),
            )
            if db.total_changes > 0: ok += 1
            else: skip += 1
        except: skip += 1
    db.commit()
    db.execute(
        "INSERT OR REPLACE INTO gov_search(rowid,title,site_name,summary) SELECT r.id,r.title,r.site_name,r.summary FROM gov_raw r WHERE r.id NOT IN (SELECT rowid FROM gov_search) AND r.site_name=?",
        (SITE_NAME,),
    )
    db.commit()
    db.close()
    print(f"  Saved: new={ok}, skip={skip}, FTS synced")

if __name__ == "__main__":
    mp = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(mp)
