#!/usr/bin/env python3
"""
crawl_zqeia.py — 肇庆市环科所环境科技有限公司 公示公告爬虫
站点: http://www.zqeia.com/h-col-101.html
CMS: 凡科(Faisco)建站
列表: JS内嵌newsList数据
详情: /h-nd-{id}.html
"""

import os, re, sys, json, time, sqlite3
import requests
import warnings
warnings.filterwarnings('ignore')

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SITE_NAME = "肇庆环科所-公示公告"
DOMAIN = "www.zqeia.com"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

stats = {"new": 0, "skip": 0, "errors": 0, "empty": 0}


def fetch_list_data():
    """从JS中提取所有栏目的newsList数据"""
    url = "http://www.zqeia.com/h-col-101.html"
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    html = r.text

    # 提取每篇文章
    items = []
    for m in re.finditer(r'{"aid":\d+,"id":(\d+),"title":"((?:\\.|[^"\\])*)","date":(\d+),', html):
        item_id = m.group(1)
        title = m.group(2)
        ts = int(m.group(3))
        if ts > 1000000000000:
            ts = ts / 1000
        date = time.strftime("%Y-%m-%d", time.localtime(ts)) if ts else ""
        url = f"http://{DOMAIN}/h-nd-{item_id}.html"
        items.append({
            "url": url,
            "title": title,
            "date": date,
            "id": item_id,
        })

    # 去重
    seen = set()
    unique = []
    for item in items:
        if item["url"] not in seen:
            seen.add(item["url"])
            unique.append(item)

    print(f"  📊 列表共 {len(unique)} 条文章")
    return unique


def extract_content(html):
    """直接用正则提取 richContent 区域"""
    m = re.search(r'<div\s+class=[\'"]richContent[^\'"]*[\'"][^>]*>(.*?)</div>\s*<!--', html, re.DOTALL)
    if m:
        c = m.group(1)
        c = re.sub(r'<(script|style)[^>]*>.*?</\1>', '', c, flags=re.DOTALL|re.I)
        c = re.sub(r'<div\s+class="[^"]*share[^"]*"[^>]*>.*?</div>', '', c, flags=re.DOTALL|re.I)
        c = re.sub(r'<div\s+class="[^"]*(?:print|tool|btn|operate)[^"]*"[^>]*>.*?</div>', '', c, flags=re.DOTALL|re.I)
        text = re.sub(r'<[^>]+>', '', c).strip()
        if len(text) > 20:
            return c.strip()
    return ""


def extract_meta(html):
    """提取标题和日期"""
    title = ""
    tm = re.search(r'<title>(.*?)<', html)
    if tm:
        title = tm.group(1).strip()

    date = ""
    dm = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    if dm:
        date = dm.group(1)
    return title, date


def insert_article(url, title, date, content):
    db = sqlite3.connect(SEARCH_DB)
    db.execute("PRAGMA journal_mode=WAL")
    try:
        db.execute(
            "INSERT OR IGNORE INTO gov_raw(site_name,source_url,page_url,title,publish_date,content,summary,status,category) VALUES(?,?,?,?,?,?,?,?,?)",
            (SITE_NAME, url, url, title, date, content, (re.sub(r'<[^>]+>', '', content)[:200] if content else title[:200]), "active", "公示公告"))
        affected = db.total_changes
        db.commit()
        if affected:
            db.execute(
                "INSERT INTO gov_search(rowid,title,site_name,summary) SELECT r.id,r.title,r.site_name,r.summary FROM gov_raw r WHERE r.page_url=? AND r.id NOT IN (SELECT rowid FROM gov_search)",
                (url,))
            db.commit()
        db.close()
        return affected
    except Exception as e:
        db.close()
        print(f"  DB error: {e}")
        return 0


def run():
    print(f"\n{'='*50}")
    print(f"🚀 {SITE_NAME}")
    print(f"{'='*50}")

    items = fetch_list_data()
    if not items:
        print("❌ 列表无数据")
        return

    new = skip = empty = errors = 0
    for i, item in enumerate(items, 1):
        for retry in range(3):
            try:
                r = requests.get(item["url"], headers=HEADERS, timeout=30)
                if r.status_code == 200:
                    r.encoding = "utf-8"
                    html = r.text
                    break
            except:
                if retry < 2:
                    time.sleep(2)
                html = None

        if not html:
            errors += 1
            continue

        pt, pd = extract_meta(html)
        real_title = pt if pt else item["title"]
        real_date = pd if pd else item["date"]

        content = extract_content(html)
        text_len = len(re.sub(r'<[^>]+>', '', content).strip()) if content else 0
        if text_len < 20:
            empty += 1
            continue

        if insert_article(item["url"], real_title, real_date, content):
            new += 1
        else:
            skip += 1

        if i % 5 == 0:
            print(f"  ...{i}/{len(items)}")
        time.sleep(0.3)

    # FTS
    print(f"\n📊 FTS同步...")
    db = sqlite3.connect(SEARCH_DB)
    try:
        db.execute("PRAGMA busy_timeout=30000")
        db.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
        db.execute("INSERT INTO gov_search(rowid,title,site_name,summary) SELECT rowid,title,site_name,summary FROM gov_raw WHERE site_name=?", (SITE_NAME,))
        db.commit()
        cnt = db.execute("SELECT COUNT(*) FROM gov_search WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
        print(f"  FTS: {cnt} 条")
    except Exception as e:
        print(f"  FTS error: {e}")
    finally:
        db.close()

    print(f"\n✅ {SITE_NAME}: 新增{new}, 跳过{skip}, 空正文{empty}, 错误{errors}")


if __name__ == "__main__":
    run()
