#!/usr/bin/env python3
"""恒光科技信息公示爬虫 — 万户网络 ezEip"""
import re, sys, os, urllib.request, ssl
sys.path.insert(0, "/root/gov_crawler")
from base_crawler import GovCrawler

SITE_NAME = "恒光科技-信息公示"
DOMAIN = "www.hgkjgf.com"
BASE = "https://www.hgkjgf.com"
LIST_TPL = "https://www.hgkjgf.com/cn/list/list_25_page_{}.html"

CTX = ssl._create_unverified_context()
HEADERS = {"User-Agent": "Mozilla/5.0"}

def http_get(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        with urllib.request.urlopen(req, context=CTX, timeout=20) as r:
            return r.read().decode('utf-8', errors='replace')
    except:
        return None

class HgkjCrawler(GovCrawler):
    def do_crawl(self):
        total_pages = 7
        for page in range(1, total_pages + 1):
            url = LIST_TPL.format(page) if page > 1 else "https://www.hgkjgf.com/cn/list/list_25.aspx"
            html = http_get(url)
            if not html:
                print(f"  ⚠️ 第{page}页获取失败")
                continue

            # Extract list items
            m = re.search(r'<div[^>]*class="NewsList"[^>]*>.*?<ul>(.*?)</ul>', html, re.S)
            if not m:
                print(f"  ⚠️ 第{page}页无列表")
                continue

            items = []
            for li in re.findall(r'<li[^>]*>(.*?)</li>', m.group(1), re.S):
                if not li.strip():
                    continue
                href_m = re.search(r"href='([^']*info_25[^']+)'", li)
                date_m = re.search(r'(\d{4}-\d{2}-\d{2})', li)
                if not href_m:
                    continue
                href = href_m.group(1)
                if not href.startswith('http'):
                    href = BASE + href
                date = date_m.group(1) if date_m else ''
                items.append((href, date))

            print(f"  📄 第{page}页: {len(items)}条")
            for href, date in items:
                detail_html = http_get(href)
                if not detail_html:
                    continue
                # Full title from <h2>
                title_m = re.search(r'<h2>(.*?)</h2>', detail_html, re.S)
                title = title_m.group(1).strip() if title_m else ''
                # Content from ContentAbout
                content_m = re.search(r'<div[^>]*class="ContentAbout"[^>]*>(.*?)</div>', detail_html, re.S)
                content = content_m.group(1).strip() if content_m else ''
                self.store_item(title=title, url=href, content=content, date=date)

    def run(self):
        self.ensure_site()
        self._stats = {"new": 0, "skip": 0, "errors": 0}
        t0 = __import__('time').time()
        self.do_crawl()
        elapsed = __import__('time').time() - t0
        print(f"\n{'─'*45}")
        print(f"✅ 新增: {self._stats['new']} | 跳过: {self._stats['skip']} | 错误: {self._stats['errors']}")
        print(f"⏱️ 耗时: {elapsed:.1f}s")
        # 直接写入主DB
        self.sync_to_local_db()

    def sync_to_local_db(self):
        """直接写入本地 /root/search.db 的 gov_raw 表"""
        import sqlite3
        db_local = self._get_db()
        rows = db_local.execute(
            "SELECT title, url, content, publish_date, summary FROM crawl_results WHERE site_id=? ORDER BY id",
            (self.site_id,),
        ).fetchall()
        db_local.close()

        if not rows:
            print("  ⏭️ 无数据")
            return

        db = sqlite3.connect("/root/search.db", timeout=30)
        added = 0
        for title, page_url, content, pub_date, summary in rows:
            try:
                db.execute(
                    "INSERT OR IGNORE INTO gov_raw (title, page_url, content, publish_date, summary, site_name) VALUES (?, ?, ?, ?, ?, ?)",
                    (title, page_url, content[:8000], pub_date or "", (summary or "")[:300], SITE_NAME),
                )
                if db.total_changes > 0:
                    added += 1
            except Exception as e:
                print(f"  DB error: {e}")
        db.commit()
        db.close()
        print(f"📤 已同步 {added}/{len(rows)} 条到 search.db")

if __name__ == '__main__':
    HgkjCrawler(
        site_name=SITE_NAME,
        domain=DOMAIN,
        db_name='crawler_results.db'
    ).run()
