#!/usr/bin/env python3
# -*- coding: utf-8 -*-
# Crawler for 安达市人民政府 - 环境保护 (环评审批)
# API: /common/search/a501a2b8e3b94ea18e9a6b580a141fc8

import requests, sqlite3, re, sys
from datetime import datetime, timedelta
import os

API_URL = "https://www.hlanda.gov.cn/common/search/a501a2b8e3b94ea18e9a6b580a141fc8?page={}&_pageSize=15&_isAgg=true&_isJson=true&_template=index&_rangeTimeGte=&_channelName="
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
SITE_NAME = "安达市环境保护（环评审批）"
FULL = "--full" in sys.argv

TITLE_PAT = re.compile(r'<meta\s+name="ArticleTitle"\s+content="([^"]+)"')
CONTENT_PAT1 = re.compile(r'<div\s+class="article_content\s+article_content_body"[^>]*>\s*<ucapcontent[^>]*>([\s\S]*?)</ucapcontent>\s*</div>', re.IGNORECASE)
CONTENT_PAT2 = re.compile(r'class="details_content"[^>]*>([\s\S]*?)</div>\s*</div>\s*</div>\s*</div>')

def fetch_page(page):
    try:
        r = requests.get(API_URL.format(page), headers=HEADERS, timeout=15)
        return r.json() if r.status_code == 200 else None
    except Exception as e:
        print(f"  API error: {e}")
        return None

def extract_detail(detail_url):
    # Fix double slash
    url = detail_url.replace("//ad/", "/ad/")
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  Detail error: {e}")
        return None, None

    tm = TITLE_PAT.search(html)
    cm = CONTENT_PAT1.search(html) or CONTENT_PAT2.search(html)

    title = tm.group(1).strip() if tm else None
    content = cm.group(1).strip() if cm else None

    return title, content

def run():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_new = total_skip = total_before = 0

    first_data = fetch_page(1)
    if not first_data:
        print("ERROR: Cannot fetch page 1")
        return

    total_elements = first_data.get("data", {}).get("total", 0)
    total_pages = (total_elements + 14) // 15
    print(f"Total: {total_elements} articles, {total_pages} pages")

    pages_to_crawl = min(total_pages, 5) if not FULL else total_pages
    print(f"Crawling {pages_to_crawl} pages")

    for pg in range(1, pages_to_crawl + 1):
        data = fetch_page(pg)
        if not data:
            print(f"Page {pg}: no data")
            continue

        articles = data.get("data", {}).get("results", [])
        if not articles:
            print(f"Page {pg}: empty")
            break

        page_new = page_skip = page_before = 0
        for art in articles:
            detail_url = art.get("url", "").strip()
            title = art.get("title", "").strip()
            release_time = art.get("publishedTimeStr", "")
            pub_date = release_time[:10] if release_time else ""

            if pub_date and pub_date < CUTOFF_DATE:
                page_before += 1
                continue

            c.execute("SELECT 1 FROM gov_raw WHERE page_url = ?", (detail_url,))
            if c.fetchone():
                page_skip += 1
                continue

            detail_title, content = extract_detail(detail_url)
            ft = detail_title or title

            if content is None:
                print(f"  Skip (no content): {ft[:40]}...")
                page_skip += 1
                continue

            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary) "
                "VALUES (?,?,?,?,?,?,?)",
                (SITE_NAME, detail_url, detail_url, ft, pub_date, content, ft)
            )
            page_new += 1
            total_new += 1

        conn.commit()
        print(f"Page {pg}: +{page_new} new, {page_skip} skip, {page_before} pre-cutoff")

        if page_before == len(articles) and pg < pages_to_crawl:
            print("All remaining before cutoff, stopping")
            break

    conn.close()
    print(f"\nDone: {total_new} new, {total_skip} skip, {total_before} before cutoff")
    return total_new, total_skip

if __name__ == "__main__":
    run()