#!/usr/bin/env python3
"""烟台经济技术开发区 - 建设项目环评审批"""
import requests, sys, os, sqlite3, re, time, json
from datetime import datetime
from bs4 import BeautifulSoup

API_URL = "https://www.yeda.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
BASE = "https://www.yeda.gov.cn"
DB = "/root/search.db"
SITE_NAME = "烟台经开区-环评审批"
TABLE = "site_yeda"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "application/json, text/plain, */*",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://www.yeda.gov.cn/col/col50341/",
}

API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "140",
    "tplSetId": "FhvjvaxoOZi0km37YAMrK",
    "pageType": "column",
    "tagId": "当前栏目列表",
    "pageId": "50341",
}

def get_conn():
    conn = sqlite3.connect(DB)
    conn.execute("CREATE TABLE IF NOT EXISTS site_yeda (id INTEGER PRIMARY KEY AUTOINCREMENT, title TEXT UNIQUE, url TEXT, date TEXT, content TEXT, summary TEXT, created_at TEXT)")
    return conn

def fetch_list(page=1, page_size=15):
    params = dict(API_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page, "pageSize": page_size}, ensure_ascii=False)
    try:
        r = requests.get(API_URL, params=params, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        data = r.json()
        if data.get("success"):
            return data["data"]["html"]
        return None
    except Exception as e:
        print("  [ERROR] fetch list page %d: %s" % (page, e))
        return None

def parse_list(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.find_all("li", class_="bt-main-r-ul-li"):
        a = li.find("a")
        span = li.find("span")
        if not a or not a.get("href"):
            continue
        title = a.get("title", "").strip()
        if not title:
            title = a.get_text(strip=True)
        href = a["href"]
        if href.startswith("/"):
            href = BASE + href
        date = span.get_text(strip=True) if span else ""
        items.append({"title": title, "url": href, "date": date})
    return items

def get_page_info(html):
    soup = BeautifulSoup(html, "html.parser")
    pagination = soup.find("div", class_="pagination")
    if pagination:
        count = int(pagination.get("count", 0))
        rows = int(pagination.get("rows", 15))
        return (count + rows - 1) // rows
    return 1

def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("  [ERROR] fetch detail %s: %s" % (url[-40:], e))
        return None

def extract_text(html):
    soup = BeautifulSoup(html, "html.parser")
    zoom = soup.find("div", class_="text", id="zoom")
    if not zoom:
        return "", ""
    text = zoom.get_text("\n", strip=True)
    text = re.sub(r"\n{3,}", "\n\n", text)
    summary = text[:300] if len(text) > 300 else text
    return text, summary

def crawl(incremental=False):
    conn = get_conn()
    cur = conn.cursor()

    existing = set()
    if incremental:
        for row in cur.execute("SELECT title FROM site_yeda"):
            existing.add(row[0])

    html = fetch_list(1)
    if not html:
        print("[ERROR] Could not fetch list")
        sys.exit(1)

    total_pages = get_page_info(html)
    print("Total pages: %d" % total_pages)

    all_items = parse_list(html)
    for p in range(2, total_pages + 1):
        html = fetch_list(p)
        if html:
            items = parse_list(html)
            all_items.extend(items)
            if p % 10 == 0 or p == total_pages:
                print("  Page %d/%d: %d items (total: %d)" % (p, total_pages, len(items), len(all_items)))
        else:
            print("  Page %d/%d: FAILED" % (p, total_pages))
        time.sleep(0.3)

    print("Total items: %d" % len(all_items))

    total_inserted = 0
    total_skipped = 0

    for item in all_items:
        title = item["title"]
        if title in existing:
            total_skipped += 1
            continue

        detail_html = fetch_detail(item["url"])
        if not detail_html:
            total_skipped += 1
            continue

        text, summary = extract_text(detail_html)
        if not text or len(text) < 10:
            total_skipped += 1
            continue

        now = datetime.now().strftime("%Y-%m-%d %H:%M:%S")
        try:
            cur.execute(
                "INSERT OR IGNORE INTO site_yeda (title, url, date, content, summary, created_at) VALUES (?,?,?,?,?,?)",
                (title, item["url"], item["date"], text, summary, now)
            )
            if cur.rowcount > 0:
                total_inserted += 1
                if total_inserted % 50 == 0:
                    print("  + [%d] %s (%s)" % (total_inserted, title[:50], item["date"]))
                    conn.commit()
            else:
                total_skipped += 1
        except Exception as e:
            print("  [ERROR] DB insert: %s" % e)

        if total_inserted % 20 == 0 and total_inserted > 0:
            conn.commit()

    conn.commit()
    conn.close()
    print("\nDone! Inserted: %d, Skipped: %d" % (total_inserted, total_skipped))

if __name__ == "__main__":
    incremental = "--incremental" in sys.argv
    requests.packages.urllib3.disable_warnings()
    crawl(incremental=incremental)
