#!/usr/bin/env python3
"""
双台子区人民政府 - 通知公告 爬虫
www.stq.gov.cn /13513/
"""

import requests
import sqlite3
import hashlib
import re
import time
from datetime import datetime, date
from bs4 import BeautifulSoup
import os

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
    'Host': 'www.stq.gov.cn',
}
BASE = 'http://175.173.244.122'
SITE_NAME = '双台子区人民政府-通知公告'
CUTOFF_DATE = date(2023, 6, 1)
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

def get_list(page):
    if page == 1:
        url = f'{BASE}/13513/'
    else:
        url = f'{BASE}/13513/list-{page}.html'
    try:
        resp = requests.get(url, headers=HEADERS, timeout=20)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return []
        soup = BeautifulSoup(resp.text, 'lxml')
        ul = soup.find('ul', class_='Q25_ListContlist')
        if not ul:
            return []
        items = []
        for li in ul.find_all('li'):
            a = li.find('a')
            spans = li.find_all('span')
            if not a or len(spans) < 2:
                continue
            href = a.get('href', '')
            title = spans[0].get_text(strip=True)
            pub_date = spans[1].get_text(strip=True)
            if not href or not title or not pub_date:
                continue
            if href.startswith('http'):
                full_url = href
            elif href.startswith('/'):
                full_url = 'http://www.stq.gov.cn' + href
            else:
                full_url = 'http://www.stq.gov.cn/' + href
            items.append({'url': full_url, 'title': title, 'pub_date': pub_date})
        return items
    except Exception as e:
        print(f'  [ERR] list page {page}: {e}')
        return []

def fetch_detail(url):
    ip_url = url.replace('http://www.stq.gov.cn', BASE)
    try:
        resp = requests.get(ip_url, headers=HEADERS, timeout=20)
        resp.encoding = 'utf-8'
        if resp.status_code != 200:
            return None
        soup = BeautifulSoup(resp.text, 'lxml')
        artcont = soup.find('div', class_='Q25_artcont')
        if artcont:
            return re.sub(r'\n\s*\n', '\n', str(artcont))
        artcont = soup.find('div', class_='Q25_articlecont')
        if artcont:
            return str(artcont)
        return None
    except:
        return None

def main():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    total_added = 0
    no_content = 0

    for page in range(1, 48):
        print(f'Page {page}/47...', end=' ', flush=True)
        items = get_list(page)
        if not items:
            print('empty')
            continue

        page_added = 0
        all_before = True
        for item in items:
            try:
                d = datetime.strptime(item['pub_date'], '%Y-%m-%d').date()
            except:
                d = date(1900, 1, 1)
            if d < CUTOFF_DATE:
                continue
            all_before = False

            # Check by page_url unique index
            c.execute('SELECT 1 FROM gov_raw WHERE page_url = ?', (item['url'],))
            if c.fetchone():
                continue

            content = fetch_detail(item['url'])
            if not content:
                no_content += 1
                content = ''

            c.execute('''INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date, content, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?)''',
                (SITE_NAME, item['url'], item['url'], item['title'],
                 item['pub_date'], content, item['title']))
            page_added += 1
            total_added += 1
            if total_added % 5 == 0:
                time.sleep(0.3)

        print(f'+{page_added}')
        conn.commit()
        if all_before and len(items) > 0:
            print(f'  All before cutoff, stopping')
            break
        time.sleep(0.5)

    conn.close()
    print(f'\nDone! Added: {total_added}, NoContent: {no_content}')

if __name__ == '__main__':
    main()
