#!/usr/bin/env python3
import os
"""瓮福（集团）有限责任公司 - 公示公告爬虫
supercache CMS, 列表页 JS 渲染, 详情页静态HTML
"""
import asyncio, re, sqlite3, sys, time, os, urllib.request
from datetime import datetime, timedelta
from playwright.async_api import async_playwright

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "瓮福集团-公示公告"
BASE = "http://www.wengfu.com"
LIST_URL = BASE + "/%E5%85%AC%E7%A4%BA%E5%85%AC%E5%91%8A"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

H = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch(url):
    try:
        req = urllib.request.Request(url, headers=H)
        resp = urllib.request.urlopen(req, timeout=30)
        return resp.read().decode('utf-8', 'ignore')
    except:
        return None

def extract_detail(html):
    title = ""
    m = re.search(r'<title>(.*?)<', html)
    if m:
        title = m.group(1).replace(" - 瓮福（集团）有限责任公司", "").strip()
    
    publish_date = ""
    m = re.search(r'发布于[：:]?\s*(\d{4}-\d{2}-\d{2})', html)
    if m:
        publish_date = m.group(1)
    
    content = ""
    # content area
    body_start = html.find('Page-content-inner')
    if body_start >= 0:
        body_html = html[body_start:]
        m = re.search(r'richtext[^>]*>(.*?)</div>', body_html, re.DOTALL)
        if m:
            content = m.group(1).strip()
            content = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', content, flags=re.DOTALL|re.I)
            content = re.sub(r'\s*(style|class|align|lang|dir|width|height|border|cellpadding|cellspacing)="[^"]*"', '', content)
    
    if not content and title:
        content = title
    
    return title, publish_date, content

async def get_articles():
    async with async_playwright() as p:
        browser = await p.chromium.launch(headless=True)
        page = await browser.new_page()
        await page.goto(LIST_URL, wait_until="networkidle", timeout=30000)
        await asyncio.sleep(3)
        
        items = await page.evaluate('''() => {
            const items = document.querySelectorAll('.cc-postslist--item');
            return Array.from(items).map(item => {
                const link = item.querySelector('a');
                const dateEl = item.querySelector('.cc-postslist--date');
                const textParts = dateEl ? dateEl.textContent.trim().split(/\\s+/) : [];
                return {
                    title: link ? link.textContent.trim() : '',
                    href: link ? link.href : '',
                    date: textParts.filter(t => /\\d{4}-\\d{2}-\\d{2}/.test(t))[0] || ''
                };
            });
        }''')
        
        await browser.close()
        return items

def run(max_pages=None):
    print(f"\n{'='*50}\n🚀 {SITE_NAME}\n{'='*50}")
    
    articles = asyncio.run(get_articles())
    print(f"📋 共 {len(articles)} 条")
    
    if not articles:
        print("  ⚠️ 无数据")
        return
    
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    new = skip = no_body = 0
    
    for i, item in enumerate(articles, 1):
        if item["date"] and item["date"] < CUTOFF:
            skip += 1
            continue
        
        html = fetch(item["href"])
        if not html:
            print(f"  ⚠️ 详情取不到: {item['title'][:30]}")
            skip += 1
            continue
        
        title, date, content = extract_detail(html)
        if not title:
            title = item["title"]
        if not date:
            date = item["date"]
        
        text_len = len(re.sub(r'<[^>]+>', '', content).strip()) if content else 0
        if text_len < 10:
            no_body += 1
            continue
        
        try:
            dr = 0
            try:
                dr = int(datetime.strptime(date, "%Y-%m-%d").timestamp())
            except:
                dr = int(time.time())
            summary = re.sub(r'<[^>]+>', '', content).strip()[:200]
            conn.execute(
                "INSERT OR IGNORE INTO gov_raw(site_name,source_url,page_url,title,publish_date,content,summary,date_rank,category) VALUES(?,?,?,?,?,?,?,?,?)",
                (SITE_NAME, item["href"], item["href"], title, date, content, summary, dr, "公示公告")
            )
            if conn.total_changes:
                new += 1
            else:
                skip += 1
        except Exception as e:
            print(f"  ❌ DB: {e}")
            skip += 1
        
        if i % 5 == 0:
            conn.commit()
            print(f"  ...{i}/{len(articles)}")
        time.sleep(0.3)
    
    conn.commit()
    # Sync FTS
    try:
        # QC20260926 去掉手写 gov_search 整站删除(抢锁源; FTS 由 gov_raw 触发器维护) 
        # conn.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
        conn.execute("INSERT OR REPLACE INTO gov_search(rowid,title,site_name,summary) SELECT rowid,title,site_name,summary FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        conn.commit()
    except Exception as e:
        print(f"⚠️ FTS: {e}")
    conn.close()
    
    # Show size info
    total = new + skip + no_body
    print(f"\n✅ {SITE_NAME}: 新增{new}, 空正文{no_body}, 跳过{skip}")

if __name__ == "__main__":
    mp = int(sys.argv[1]) if len(sys.argv) > 1 and sys.argv[1].isdigit() else None
    run(mp)
