#!/usr/bin/env python3
"""全椒县生态环境分局 - 信息公开爬虫
API: czxxgk/site/label/8888 (JSON, 需 Referer + X-Requested-With)
"""
import re, time, os, sqlite3
from datetime import datetime, date
from concurrent.futures import ThreadPoolExecutor, as_completed
import requests

BASE = "https://www.quanjiao.gov.cn"
API_URL = "https://www.quanjiao.gov.cn/czxxgk/site/label/8888"
CUTOFF_DATE = date(2023, 6, 17)
DB_PATH = os.environ.get("DB_PATH", "/root/search.db")
SITE_NAME = "全椒县-生态环境分局"
MAX_WORKERS = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "application/json, text/javascript, */*; q=0.01",
    "Accept-Language": "zh-CN,zh;q=0.9",
    "Referer": "https://www.quanjiao.gov.cn/public/column/161055016?type=4&action=list&nav=3",
    "X-Requested-With": "XMLHttpRequest",
    "Connection": "keep-alive",
}


def fetch_api(page):
    for i in range(3):
        try:
            r = requests.get(API_URL, params={
                "labelName": "publicInfoList", "siteId": 2653861,
                "organId": 161055016, "type": 4, "dateFormat": "yyyy-MM-dd",
                "isJson": "true", "pageIndex": page, "pageSize": 20
            }, timeout=30, headers=HEADERS)
            if r.status_code == 200 and r.text:
                return r.json()
        except Exception as e:
            if i < 2:
                time.sleep(3)
    return None


def fetch(url):
    h = HEADERS.copy()
    h["Referer"] = "https://www.quanjiao.gov.cn/public/column/161055016?type=4&action=list&nav=3"
    h.pop("X-Requested-With", None)
    h.pop("Accept", None)
    for i in range(3):
        try:
            r = requests.get(url, timeout=30, headers=h)
            r.encoding = "utf-8"
            if r.status_code == 200 and r.text:
                return r.text
        except:
            if i < 2:
                time.sleep(3)
    return None


def fetch_detail(detail_url):
    html = fetch(detail_url)
    if not html:
        return (detail_url, None, None, None)
    title = ""
    content = ""
    pub_date = ""
    m = re.search(r'<meta name="ArticleTitle" content="([^"]*)"', html)
    if m:
        title = m.group(1).strip()
    m = re.search(r'<meta name="PubDate" content="([^"]*)"', html)
    if m:
        pub_date = m.group(1).strip()[:10]
    m = re.search(r'<div class="gkwz_contnet[^"]*"[^>]*>(.*?)</div>\s*</div>\s*</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    if not content:
        m2 = re.search(r'<div class="clearfix xxgkcontent[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
        if m2:
            content = m2.group(1).strip()
    return (detail_url, title, content, pub_date)


def main():
    print("=== 全椒县-生态环境分局 ===", flush=True)
    print("获取列表...", flush=True)
    all_items = []
    for page in range(1, 31):
        data = fetch_api(page)
        if not data or not data.get("data"):
            print(f"  第{page}页: 无数据或API限制", flush=True)
            break
        items = data["data"]
        filtered = []
        for item in items:
            pd = item.get("publishDate", "")[:10]
            if pd:
                d = date.fromisoformat(pd)
                if d >= CUTOFF_DATE:
                    filtered.append({
                        "title": item.get("title", ""),
                        "date": pd,
                        "url": f"{BASE}/public/161055016/{item.get('contentId','')}.html",
                        "contentId": item.get("contentId", "")
                    })
        all_items.extend(filtered)
        last_date = items[-1].get("publishDate", "")[:10] if items else ""
        print(f"  第{page}页: {len(items)}条, 近3年{len(filtered)}条, 最晚{last_date}", flush=True)
        if items and not filtered and page > 1:
            break
        time.sleep(1)

    print(f"  总计: {len(all_items)}条", flush=True)
    if not all_items:
        return

    conn = sqlite3.connect(DB_PATH)
    cursor = conn.cursor()
    total_new = 0
    done = 0

    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        fut_map = {executor.submit(fetch_detail, it["url"]): it for it in all_items}
        for fut in as_completed(fut_map):
            item = fut_map[fut]
            url, title, content, pub_date = fut.result()
            done += 1
            if done % 50 == 0:
                print(f"  详情 {done}/{len(all_items)}...", flush=True)
            if title is None:
                continue
            if not title:
                title = item["title"]
            if not pub_date:
                pub_date = item["date"]
            if not content:
                content = title
            try:
                cursor.execute("""
                    INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, content, publish_date, summary, date_rank)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (SITE_NAME, title, url, content, pub_date, content,
                      int(datetime.strptime(pub_date, "%Y-%m-%d").timestamp())))
                if cursor.rowcount > 0:
                    total_new += 1
            except:
                pass
            if done % 100 == 0:
                conn.commit()

    conn.commit()
    cursor.execute("SELECT COUNT(*), MIN(publish_date), MAX(publish_date) FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    cnt, min_d, max_d = cursor.fetchone()
    conn.close()

    print(f"\n=== 完成 ===", flush=True)
    print(f"  新增: {total_new}条", flush=True)
    print(f"  累计: {cnt}条 ({min_d} ~ {max_d})", flush=True)


if __name__ == "__main__":
    main()
