#!/usr/bin/env python3
import os
"""
Crawler for 齐齐哈尔市龙沙区 - 环境保护
https://www.qqhrlsq.gov.cn/lsq/c100747/zfxxgk_list.shtml
API: GET /search/{channelId}?page=N&_pageSize=10&_isJson=true
"""

import json, sys, time, sqlite3, re
from urllib.request import urlopen, Request

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = 'https://www.qqhrlsq.gov.cn'
SITE_NAME = '齐齐哈尔市龙沙区-环境保护'
CHANNEL_ID = '7263f408df684117b349c7a880d4b8d7'
PAGE_SIZE = 10
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

conn = sqlite3.connect(DB_PATH)
conn.execute("PRAGMA busy_timeout=5000")


def fetch_api(page):
    url = f'{BASE_URL}/search/{CHANNEL_ID}?page={page}&_pageSize={PAGE_SIZE}&_isJson=true'
    req = Request(url, headers=HEADERS)
    resp = urlopen(req, timeout=30)
    return json.loads(resp.read())


def fetch_detail(url):
    req = Request(url, headers=HEADERS)
    resp = urlopen(req, timeout=30)
    html = resp.read().decode('utf-8')
    c = re.search(r'<div[^>]*class=[\"\']article-content[\"\']?[^>]*>(.*)', html, re.DOTALL)
    if c:
        # Use depth counting to find the closing </div>
        rest = c.group(1)
        depth = 1
        pos = 0
        for i, ch in enumerate(rest):
            if ch == '<':
                tag_end = rest.find('>', i)
                if tag_end == -1:
                    break
                tag = rest[i+1:tag_end].strip()
                tag_name = tag.split()[0]
                if tag.startswith('/'):
                    depth -= 1
                    if depth == 0:
                        pos = tag_end + 1
                        break
                elif not tag.startswith('!--') and tag_name not in ['br', 'hr', 'img', 'input', 'meta', 'link', '!DOCTYPE'] and not tag.endswith('/'):
                    depth += 1
                i = tag_end
        content = rest[:pos-1].strip() if pos > 0 else rest.strip()
    else:
        content = ''
    content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
    content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
    content = content.strip()
    return content


def save(title, content, pub_date, url):
    if not title:
        return False
    title = title.strip()
    if not title:
        return False
    conn.execute(
        'INSERT OR REPLACE INTO gov_raw (title, content, publish_date, page_url, source_url, site_name) VALUES (?, ?, ?, ?, ?, ?)',
        (title, content or '', pub_date, url, url, SITE_NAME)
    )
    return True


def main():
    start = time.time()
    total_saved = 0
    total_articles = 0

    print(f'[qqhrlsq] Fetching API page 1...')
    raw = fetch_api(1)
    total = raw.get('data', {}).get('total', 0)
    total_pages = (total + PAGE_SIZE - 1) // PAGE_SIZE
    print(f'[qqhrlsq] Total: {total} articles, {total_pages} pages')

    for page in range(1, total_pages + 1):
        if page == 1:
            raw_page = raw
        else:
            print(f'[qqhrlsq] Fetching page {page}/{total_pages}...')
            raw_page = fetch_api(page)
            time.sleep(0.15)

        results = raw_page.get('data', {}).get('results', [])
        for art in results:
            title = art.get('title', '')
            api_content = art.get('content', '')
            pub_date = art.get('publishedTimeStr', '')
            article_url = art.get('url', '')

            total_articles += 1

            if not article_url:
                continue

            # Try to get detail content
            content = ''
            try:
                content = fetch_detail(article_url)
                time.sleep(0.15)
            except Exception as e:
                print(f'  [warn] Detail fetch failed: {e}')

            if save(title, content, pub_date, article_url):
                total_saved += 1

        conn.commit()
        print(f'[qqhrlsq] Progress: {total_saved}/{total} saved')

    elapsed = time.time() - start
    print(f'[qqhrlsq] Done: {total_saved}/{total_articles} saved in {elapsed:.1f}s')
    conn.close()


if __name__ == '__main__':
    main()
