#!/usr/bin/env python3
import os
"""
crawl_rudong.py - 如东县洋口港经济开发区-公告公示
TrueCMS，API分页+requests取详情
"""
import os, re, sys, json, time, sqlite3
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
import requests

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "如东县洋口港经济开发区-公告公示"
CATEGORY = "环保公示"
BASE = "https://www.rudong.gov.cn"
LIST_API = BASE + "/truecms/messageController/getMessage.do"
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36', 'Referer': BASE + '/ykgjjkfq/gggs/gggs.html'}
COOKIES = {'JSESSIONID': 'hermes_crawler'}
PER_PAGE = 10
COLUMN_ID = '0d8c9378-205b-4f2f-84dd-2411bbb1fed7'

print(f"[rudong] 3-year cutoff: {CUTOFF_DATE}")

# Get total count
resp = requests.get(LIST_API, params={'columnId': COLUMN_ID, 'startrecord': 0, 'endrecord': 1, 'perPage': PER_PAGE},
                     headers=HEADERS, cookies=COOKIES, timeout=30)
data = resp.json()
html_data = data.get('result', '')
total_match = re.search(r'<totalrecord>(\d+)</totalrecord>', html_data)
total_count = int(total_match.group(1)) if total_match else 0
print(f"[rudong] Total records: {total_count}")

total_pages = (total_count + PER_PAGE - 1) // PER_PAGE
print(f"[rudong] Total pages: {total_pages}")

conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()
new_count = dup_count = error_count = 0
all_items = []

for pg in range(1, total_pages + 1):
    start = (pg - 1) * PER_PAGE
    end = min(start + PER_PAGE, total_count)
    
    params = {'columnId': COLUMN_ID, 'startrecord': start, 'endrecord': end, 'perPage': PER_PAGE}
    try:
        resp = requests.get(LIST_API, params=params, headers=HEADERS, cookies=COOKIES, timeout=30)
        data = resp.json()
        html_data = data.get('result', '')
    except Exception as e:
        print(f"  [WARN] Page {pg} error: {e}")
        continue
    
    items = re.findall(
        r'<a[^>]+href="(/ykgjjkfq/gggs/content/[^"]+)"[^>]*>(.*?)</a>.*?<span>(.*?)</span>',
        html_data, re.DOTALL
    )
    for href, title, date_str in items:
        title = BeautifulSoup(title, 'html.parser').get_text(strip=True)
        all_items.append((title, href.strip(), date_str.strip()))
    
    print(f"  Page {pg}: {len(items)} items")
    time.sleep(0.5)

print(f"[rudong] Total items extracted: {len(all_items)}")

# Filter by cutoff
active = [(t, h, d) for t, h, d in all_items if d >= CUTOFF_DATE]
skipped = len(all_items) - len(active)
print(f"[rudong] Within 3yr: {len(active)}, older: {skipped}")

for idx, (title, href, date_str) in enumerate(active, 1):
    detail_url = href if href.startswith('http') else BASE + href
    
    cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
    if cur.fetchone():
        dup_count += 1
        continue
    
    try:
        resp = requests.get(detail_url, headers=HEADERS, timeout=20)
        resp.encoding = 'utf-8'
        html = resp.text
    except Exception as e:
        error_count += 1
        if idx <= 3:
            print(f"  [WARN] Fetch error: {detail_url[:60]} -> {e}")
        continue
    
    d_title = title
    d_date = date_str
    body = ''
    
    mt = re.search(r'ArticleTitle"\s*content="([^"]+)"', html)
    if mt: d_title = mt.group(1).strip()
    
    md = re.search(r'PubDate"\s*content="([^"]+)"', html)
    if md:
        m = re.search(r'(\d{4}-\d{2}-\d{2})', md.group(1))
        if m: d_date = m.group(1)
    
    if d_date < CUTOFF_DATE:
        continue
    
    cm = re.search(r'class="cont[^"]*"[^>]*>(.*?)</div>\s*</div>\s*</div>', html, re.DOTALL)
    if cm:
        body = cm.group(1).strip()
    if not body:
        body_match = re.search(r'class="cont[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
        if body_match:
            body = body_match.group(1).strip()
    if not body:
        error_count += 1
        continue
    
    date_rank = int(d_date.replace("-", "")) if d_date else 0
    text_soup = BeautifulSoup(body, 'html.parser')
    summary = text_soup.get_text(strip=True)[:200]
    
    cur.execute(
        "INSERT OR IGNORE INTO gov_raw "
        "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
        "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
        (SITE_NAME, d_title, detail_url, d_date, body, date_rank, CATEGORY, summary)
    )
    if cur.rowcount > 0:
        new_count += 1
    
    conn.commit()
    if idx % 20 == 0 or idx == len(active):
        print(f"  Progress {idx}/{len(active)}: +{new_count} new, {dup_count} dup, {error_count} err")
    
    time.sleep(0.3)

conn.close()
print(f"\n[rudong] Summary: +{new_count} new, {dup_count} dup, {error_count} err")
print(json.dumps({"site": SITE_NAME, "new": new_count, "dup": dup_count, "err": error_count}, ensure_ascii=False))
