#!/usr/bin/env python3
import os
"""
crawl_yongxiu2.py - 永修县公共企事业单位信息公开平台-通知公告
https://www.yongxiu.gov.cn/xzwzx/ztzl/yxxggqsypt/tzgg/
自定义CMS, 16页×15条/页≈240条(3年cutoff内)
"""
import os, re, sys, time, urllib.request, urllib.error
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "https://www.yongxiu.gov.cn/xzwzx/ztzl/yxxggqsypt/tzgg"
LIST_URL = BASE_URL + "/index.html"
HEADERS = {'User-Agent': 'Mozilla/5.0'}

# _MAX_PG support
_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == '--pages' and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break

CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
print(f"[yx2] 3-year cutoff: {CUTOFF}")

def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=20).read().decode('utf-8', errors='replace')
    except Exception as e:
        print(f"[yx2] fetch fail {url}: {e}")
        return ""

def get_detail(url):
    """Extract title, date, body from detail page."""
    html = fetch(url)
    if not html:
        return None, None, None
    soup = BeautifulSoup(html, 'html.parser')
    
    # title
    title = ''
    mt = soup.find('p', class_='mainTitle')
    if mt: title = mt.get_text(strip=True)
    
    # date: "发布日期：2026-06-18 14:14"
    date_str = ''
    sp = soup.find('span', class_='span1')
    if sp:
        txt = sp.get_text()
        m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
        if m: date_str = m.group(1)
    
    # body: div#content.xl-text
    body_el = soup.find('div', id='content', class_='xl-text')
    body = ''
    if body_el:
        body = ''.join(str(c) for c in body_el.children).strip()
    
    return title, date_str, body

# --- get total pages ---
html = fetch(LIST_URL)
page_match = re.search(r'createPage\((\d+)', html)
if not page_match:
    # try createPageHTML
    page_match = re.search(r'createPageHTML\((\d+)', html)

if not page_match:
    print("[yx2] Cannot find page count!")
    sys.exit(1)
total_pages = int(page_match.group(1))
print(f"[yx2] Total pages: {total_pages}")
if _MAX_PG and total_pages > _MAX_PG:
    total_pages = _MAX_PG
    print(f"[yx2] Limited to {_MAX_PG} pages via --pages")

# --- crawl ---
import sqlite3
conn = sqlite3.connect(DB_PATH, timeout=30)
cur = conn.cursor()

total_new = 0
total_skip = 0

for page_idx in range(total_pages):
    if page_idx == 0:
        list_url = LIST_URL
    else:
        list_url = BASE_URL + f"/index_{page_idx}.html"
    
    html = fetch(list_url)
    if not html:
        continue
    
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_=re.compile(r'doc_list'))
    if not ul:
        print(f"[yx2] Page {page_idx+1}: no doc_list")
        continue
    
    items = [li for li in ul.find_all('li', recursive=False) if li.find('a')]
    print(f"[yx2] Page {page_idx+1}/{total_pages}: {len(items)} items", end='')
    
    for li in items:
        a = li.find('a')
        href = a.get('href', '')
        title = a.get('title', a.get_text(strip=True))
        
        # date from list
        date_span = li.find('span', class_='date')
        date_str = date_span.get_text(strip=True) if date_span else ''
        
        # filter by date
        if date_str and date_str < CUTOFF:
            continue
        
        # build absolute URL
        if href.startswith('./'):
            detail_url = BASE_URL + '/' + href[2:]
        elif href.startswith('/'):
            detail_url = 'https://www.yongxiu.gov.cn' + href
        elif not href.startswith('http'):
            detail_url = BASE_URL + '/' + href
        else:
            detail_url = href
        
        # get detail
        d_title, d_date, body = get_detail(detail_url)
        if not d_title: d_title = title
        if not d_date and date_str: d_date = date_str
        
        site_name = "永修县公共企事业单位-通知公告"
        
        # check existing
        cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, site_name))
        if cur.fetchone():
            total_skip += 1
            continue
        
        body_clean = body if body else ''
        cur.execute(
            "INSERT OR IGNORE INTO gov_raw (title, content, page_url, site_name, publish_date) VALUES (?,?,?,?,?)",
            (d_title, body_clean, detail_url, site_name, d_date)
        )
        if cur.rowcount > 0:
            total_new += 1
            if total_new % 15 == 0:
                conn.commit()
    
    print(f" → new:{total_new} skip:{total_skip}")
    conn.commit()
    time.sleep(0.3)

conn.commit()
conn.close()
print(f"[yx2] DONE: {total_new} new, {total_skip} existing skipped")
