#!/usr/bin/env python3
"""crawl_hipac.py — 江苏淮安工业园区管委会 公告公示爬虫"""
import os, re, sys, time, json, sqlite3, subprocess
import requests, warnings
warnings.filterwarnings('ignore')
DOMAIN = "hipac.huaian.gov.cn"
DOMAIN = "hipac.huaian.gov.cn"

BASE_DIR = os.path.dirname(os.path.abspath(__file__))
SITE_NAME = "淮安工业园区"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
API_URL = "http://hipac.huaian.gov.cn/articleCommonController/lists.do"
DETAIL_DOMAIN = "https://hipac.huaian.gov.cn"
TOPIC = "5632"
RDEPTID = "0000000064a8f16d0164ae1f25dd00ff"
TOTAL_PAGES = 147
PAGE_SIZE = 10

stats = {"new": 0, "skip": 0, "errors": 0}

def init_db():
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT, site_name TEXT, source_url TEXT,
        page_url TEXT, title TEXT, publish_date TEXT, date_rank INTEGER DEFAULT 0,
        summary TEXT, status TEXT, category TEXT DEFAULT '', visits INTEGER DEFAULT 0,
        content TEXT DEFAULT '', tags TEXT DEFAULT ''
    )''')
    try:
        conn.execute('''CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(
            title, content, site_name,
            content='gov_raw', content_rowid='id', tokenize='unicode61'
        )''')
    except sqlite3.OperationalError:
        pass
    conn.commit()
    conn.close()

def store_record(title, url, content, date, summary):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        conn.execute("INSERT OR IGNORE INTO gov_raw (title, content, publish_date, page_url, source_url, site_name, summary, tags) VALUES (?,?,?,?,?,?,?,?)",
                     (title, content, date, url, DOMAIN, SITE_NAME, summary, ""))
        if conn.total_changes > 0:
            row_id = conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (url,)).fetchone()
            if row_id:
                try:
                    conn.execute("INSERT OR REPLACE INTO gov_search(rowid, title, site_name) VALUES (?,?,?,?)",
                                 (row_id[0], title, content, SITE_NAME))
                except sqlite3.IntegrityError:
                    pass
            stats["new"] += 1
        else:
            stats["skip"] += 1
    except:
        stats["errors"] += 1
    finally:
        conn.close()


def fetch_list_page(session, page_num):
    data = {"topic": TOPIC, "page": str(page_num), "pagesize": str(PAGE_SIZE), "title": "", "rdeptid": RDEPTID}
    headers = {"User-Agent": "Mozilla/5.0", "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
               "Referer": "http://hipac.huaian.gov.cn/cmsweb/zwgk/gyyq/index.html?topic=5632", "X-Requested-With": "XMLHttpRequest"}
    try:
        r = session.post(API_URL, data=data, headers=headers, timeout=30, verify=False)
        items = r.json().get("value", {}).get("list", [])
        result = []
        for item in items:
            title = item.get("title", "").strip().replace("\r\n", " ").replace("\n", " ")
            if not title: continue
            path = item.get("path", "")
            if not path: continue
            result.append({"url": f"{DETAIL_DOMAIN}/{path}", "title": title, "date": (item.get("releaseTime") or "")[:10]})
        return result
    except Exception as e:
        print(f"  ⚠️ API错误(第{page_num}页): {e}")
        return None

def fetch_detail(session, url):
    for attempt in range(3):
        try:
            r = session.get(url, timeout=30, verify=False)
            html = r.text
            if len(html) < 500:
                time.sleep(2); continue

            # 标题（ArticleTitle meta优先，不含站点名前缀）
            title = ""
            m = re.search(r'<meta[^>]*name="ArticleTitle"[^>]*content="([^"]+)"', html, re.I)
            if m:
                title = m.group(1).strip().replace("\r\n", " ").replace("\n", " ")
            if not title:
                m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
                if m:
                    title = m.group(1).strip().replace("\r\n", " ").replace("\n", " ")
                    if " - " in title:
                        title = title.split(" - ", 1)[-1]

            # 正文
            content = ""
            m = re.search(r'<div[^>]*class="content_body"[^>]*>(.*?)</div>\s*</div>\s*</div>', html, re.DOTALL)
            if not m:
                m = re.search(r'<div[^>]*class="content_d"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
            if m:
                c = m.group(1).strip()
                c = re.sub(r'<script[^>]*>.*?</script>', '', c, flags=re.DOTALL|re.I)
                c = re.sub(r'<style[^>]*>.*?</style>', '', c, flags=re.DOTALL|re.I)
                content = c

            # 日期
            date = ""
            m = re.search(r'公开日期[：:]\s*(\d{4}-\d{2}-\d{2})', html)
            if m: date = m.group(1)

            # 摘要
            summary = re.sub(r'<[^>]+>', '', content or '')
            summary = re.sub(r'\s+', ' ', summary).strip()[:500] if summary else title
            return {"title": title, "content": content, "date": date, "summary": summary}
        except Exception as e:
            if attempt < 2: time.sleep(2)
    return None

def process_item(session, item):
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    exists = conn.execute("SELECT 1 FROM gov_raw WHERE page_url=? AND site_name=?", (item["url"], SITE_NAME)).fetchone()
    conn.close()
    if exists: return False
    print(f"  📄 {item['title'][:40]}...", end="", flush=True)
    d = fetch_detail(session, item["url"])
    if not d:
        print(" ✗ 详情失败"); stats["errors"] += 1; return False
    store_record(d["title"] or item["title"], item["url"], d["content"], d.get("date") or item.get("date", ""), d["summary"])
    print(f" ✅ {len(d.get('content',''))}B")
    return True

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--full", action="store_true")
    parser.add_argument("--sync", action="store_true")
    parser.add_argument("--test", type=int, default=0)
    args = parser.parse_args()
    init_db()
    if args.sync: print("--sync 已弃用，脚本直接写 search.db"); return

    s = requests.Session()
    s.headers.update({"User-Agent": "Mozilla/5.0"})
    s.get("http://hipac.huaian.gov.cn/cmsweb/zwgk/gyyq/index.html?topic=5632", timeout=30, verify=False)

    start = time.time()
    total_new = 0
    max_pages = TOTAL_PAGES if args.full else 1

    for pn in range(1, max_pages + 1):
        if args.test and total_new >= args.test: break
        items = fetch_list_page(s, pn)
        if not items: break
        print(f"📃 第{pn}页 ({len(items)}条)")
        for item in items:
            if args.test and total_new >= args.test: break
            if process_item(s, item): total_new += 1
            time.sleep(0.5)

    elapsed = time.time() - start
    print(f"\n{'='*50}\n🏁 新增:{stats['new']} 跳过:{stats['skip']} 错误:{stats['errors']}\n⏱️ {elapsed:.0f}s")
    if stats["new"] > 0: pass  # 直接写 search.db，无需同步

if __name__ == "__main__":
    main()
