#!/usr/bin/env python3
"""廉江开发区管委会-公示公告 crawler"""
import os, sys, re, requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta

BASE_URL = "http://www.lianjiang.gov.cn/qtlm/yqlj/ljzfbm/gdljjjkfqglwyh/gsgg/gsgg"
SITE_NAME = "廉江开发区管委会-公示公告"
DATE_THRESHOLD = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
import urllib3
urllib3.disable_warnings()

def fetch_page(url):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    return r.text

def parse_list(html):
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='listTit')
    if not ul:
        return []
    items = []
    for li in ul.find_all('li'):
        a = li.find('a')
        span = li.find('span')
        if not a:
            continue
        href = a.get('href', '')
        if not href.startswith('http'):
            href = 'http://www.lianjiang.gov.cn' + href
        title = a.get_text(strip=True)
        # Remove trailing date from title if present
        title = re.sub(r'\d{4}-\d{2}-\d{2}$', '', title).strip()
        date_str = span.get_text(strip=True) if span else ''
        items.append({'title': title, 'url': href, 'date': date_str})
    return items

def get_total_pages():
    html = fetch_page(BASE_URL + '/')
    # Find max page from "最后一页" link
    m = re.search(r'index_(\d+)\.html">最后一页<', html)
    if m:
        return int(m.group(1))
    # Fallback: count pagination links
    pages = re.findall(r'index_(\d+)\.html', html)
    return max(int(p) for p in pages) if pages else 1

def fetch_detail(url):
    html = fetch_page(url)
    soup = BeautifulSoup(html, 'html.parser')
    # Full title from <title> tag, strip site suffix
    full_title = ''
    if soup.title:
        t = soup.title.get_text(strip=True)
        # Remove " - 廉江市人民政府门户网站" suffix
        t = re.sub(r'\s*[-–—]\s*廉江市人民政府门户网站\s*$', '', t).strip()
        full_title = t
    cons = soup.find('div', class_='showCons')
    if cons:
        return str(cons), full_title
    return '', full_title

def main():
    import sqlite3
    
    total_pages = get_total_pages()
    print(f"共 {total_pages} 页")
    
    all_items = []
    for page in range(1, total_pages + 1):
        url = BASE_URL + '/' if page == 1 else f'{BASE_URL}/index_{page}.html'
        html = fetch_page(url)
        items = parse_list(html)
        if not items:
            print(f"Page {page}: 0 items, stopping")
            break
        all_items.extend(items)
        print(f"Page {page}/{total_pages}: {len(items)} items (累计 {len(all_items)})")
    
    # Filter by date
    to_fetch = [it for it in all_items if it['date'] >= DATE_THRESHOLD]
    print(f"\n近3年: {len(to_fetch)}/{len(all_items)} 条")
    
    # Fetch details
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        site_name TEXT, page_url TEXT UNIQUE, title TEXT,
        content TEXT, publish_date TEXT, summary TEXT,
        created_at TEXT DEFAULT (datetime('now','localtime'))
    )''')
    
    new_count = 0
    for i, item in enumerate(to_fetch, 1):
        content, full_title = fetch_detail(item['url'])
        if content:
            title_to_use = full_title if full_title else item['title']
            try:
                conn.execute(
                    'INSERT OR IGNORE INTO gov_raw (site_name, page_url, title, content, publish_date, summary) VALUES (?,?,?,?,?,?)',
                    (SITE_NAME, item['url'], title_to_use, content, item['date'], SITE_NAME)
                )
                if conn.total_changes > new_count:
                    new_count = conn.total_changes
                conn.commit()
            except Exception as e:
                print(f"  DB error: {e}")
        if i % 10 == 0:
            print(f"  详情 {i}/{len(to_fetch)}...")
    
    # Count results
    before_count = conn.execute('SELECT COUNT(*) FROM gov_raw WHERE site_name=?', (SITE_NAME,)).fetchone()[0]
    print(f"\n=== 完成 ===")
    print(f"已入库: {before_count}")
    conn.close()

if __name__ == '__main__':
    main()
