#!/usr/bin/env python3
"""
crawl_jl_gov.py -- Jiangling County Government Announcements
Site: http://zwgk.jiangling.gov.cn/list.shtml?column_id=7691
CMS: Custom (Jingzhou News Network), POST API for list data
"""

import sys, re, time, os, json
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = '江陵县人民政府-公告信息'
DOMAIN = "zwgk.jiangling.gov.cn"
API_URL = "http://zwgk.jiangling.gov.cn/api/content_center/document/list-data"
DEPT_ID = 6
COLUMN_ID = 7691
PAGE_SIZE = 15

THREE_YEARS_AGO = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
stats = {"new": 0, "skip": 0, "errors": 0}

session = requests.Session()
session.headers.update({
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Content-Type": "application/json;charset=UTF-8"
})


def fetch_list(page_index):
    try:
        r = session.post(API_URL, json={
            "dept_id": DEPT_ID,
            "column_id": COLUMN_ID,
            "has_related": False,
            "page_index": page_index,
            "page_size": PAGE_SIZE,
            "randNumber": time.time()
        }, timeout=30)
        if r.status_code == 200:
            data = r.json()
            if data.get("code") == 200:
                return data["data"]["list"]
        return []
    except Exception as e:
        print(f"  [ERROR] API failed: {e}")
        return []


def fetch_detail(url):
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        if r.status_code != 200:
            return "", ""
        html_text = r.text
    except Exception:
        return "", ""

    soup = BeautifulSoup(html_text, 'html.parser')

    # Date from meta tag
    meta_date = soup.find('meta', attrs={"name": "PubDate"})
    date_str = ''
    if meta_date and meta_date.get('content'):
        date_str = meta_date['content'][:10]

    # Content
    content_div = soup.find('div', id='mainText')
    if not content_div:
        content_div = soup.find('div', class_='doc_text')

    content_html = ""
    if content_div:
        content_html = str(content_div)

    return content_html, date_str


def store_item(title, url, content, date_str):
    try:
        import sqlite3
        conn = sqlite3.connect(DB_PATH, timeout=60)
        c = conn.cursor()
        summary = BeautifulSoup(content, 'html.parser').get_text(separator=' ', strip=True)[:500] if content else ''
        c.execute("INSERT OR IGNORE INTO gov_raw (site_name, title, page_url, publish_date, content, source_url, summary, category) VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                  (SITE_NAME, title, url, date_str, content, url, summary, "tzgg"))
        if c.rowcount > 0:
            stats["new"] += 1
        else:
            stats["skip"] += 1
        conn.commit()
        conn.close()
    except Exception as e:
        print(f"  [ERROR] DB: {e}")
        stats["errors"] += 1


def crawl(full=False, test_n=None):
    print("=" * 60)
    print("Jiangling County Government Announcements crawler")
    print("=" * 60)

    seen = 0
    page = 1

    while True:
        print(f"\n--- Page {page} ---")
        items = fetch_list(page)
        if not items:
            print("  No more items")
            break

        print(f"  Got {len(items)} items ({items[0]['rel_time'][:10]} ~ {items[-1]['rel_time'][:10]})")

        for idx, it in enumerate(items):
            title = it.get('title', '')
            pub_url = it.get('pub_url', '')
            rel_time = it.get('rel_time', '')[:10]

            if rel_time and rel_time < THREE_YEARS_AGO:
                print(f"  [SKIP] Too old: {rel_time} {title[:30]}")
                stats["skip"] += 1
                continue

            print(f"  [{idx+1}] {title[:40]} | {rel_time}")

            if test_n and seen >= test_n:
                print(f"\n  Test mode: {test_n} items done")
                return

            content, detail_date = fetch_detail(pub_url)
            date_str = detail_date or rel_time
            store_item(title, pub_url, content, date_str)
            seen += 1
            time.sleep(0.5)

        # Check if we've passed the cutoff
        items_date = items[-1]['rel_time'][:10]
        if items_date < THREE_YEARS_AGO:
            print(f"  Reached 3-year cutoff ({items_date} < {THREE_YEARS_AGO})")
            break

        page += 1
        if page > 60:
            break

    print(f"\nDone! New: {stats['new']}, Skip: {stats['skip']}, Errors: {stats['errors']}")


if __name__ == '__main__':
    full = '--full' in sys.argv
    test_n = None
    for arg in sys.argv:
        if arg.startswith('--test='):
            test_n = int(arg.split('=')[1])
    crawl(full=full, test_n=test_n)
