#!/usr/bin/env python3
"""陕州区人民政府-公告公示 爬虫"""
import re, os, sqlite3
from datetime import datetime, timezone, timedelta

BASE = 'https://www.shanzhou.gov.cn'
SITE_NAME = '陕州公告公示'
DB = os.getenv("SEARCH_DB", "/root/search.db")
LIST_URL = BASE + '/18019/0000/zhengfuxinxi-{page}.html'

tz = timezone(timedelta(hours=8))
cutoff = datetime.now(tz) - timedelta(days=365*3)
print(f'Cutoff: {cutoff.isoformat()}')

count = 0
page = 1
while True:
    url = LIST_URL.format(page=page)
    html = os.popen(f'curl -s "{url}"').read()
    if not html.strip():
        print(f'Page {page}: empty, stop')
        break
    links = re.findall(r'href=["\'](https?://www\.shanzhou\.gov\.cn/18019/[^"\']+)["\']', html)
    dates_in_list = re.findall(r'<span class="hous fr">(\d{4}-\d{2}-\d{2})</span>', html)
    if not links:
        print(f'Page {page}: no links, stop')
        break
    print(f'Page {page}: {len(links)} links')
    all_before = True
    for i, link in enumerate(links):
        list_date = dates_in_list[i] if i < len(dates_in_list) else ''
        if list_date < cutoff.strftime('%Y-%m-%d'):
            continue
        all_before = False
        detail_html = os.popen(f'curl -sL "{link}"').read()
        if not detail_html or len(detail_html) < 500:
            continue
        title = ''
        m = re.search(r'<title>(.*?)</title>', detail_html, re.DOTALL)
        if m:
            title = re.sub(r'\s*-\s*公告公示\s*-\s*陕州区人民政府网站\s*|\s*-\s*陕州区人民政府网站\s*', '', m.group(1)).strip()
        pub_date = list_date
        if not pub_date:
            m = re.search(r'发布日期[：:]<[^>]*>(\d{4}-\d{2}-\d{2})', detail_html)
            if m: pub_date = m.group(1)
        if not pub_date:
            continue
        content_html = ''
        m = re.search(r'<div class="content clearfix"[^>]*>(.*?)</div>', detail_html, re.DOTALL)
        if m:
            content_html = m.group(1).strip()
        if not content_html:
            continue
        content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL)
        content_html = re.sub(r'<style[^>]*>.*?</style>', '', content_html, flags=re.DOTALL)
        content_html = content_html.strip()
        if not content_html or len(content_html) < 20:
            continue
        try:
            conn = sqlite3.connect(DB, timeout=60)
            c = conn.cursor()
            c.execute('INSERT OR REPLACE INTO gov_raw (id, title, content, publish_date, source_url, page_url, site_name, summary, script_name) VALUES (?,?,?,?,?,?,?,?, \'crawl_shanzhou.py\')',
                      (None, title, content_html, pub_date, link, link, SITE_NAME, title[:200]))
            conn.commit()
            conn.close()
            count += 1
        except Exception as e:
            print(f'  DB error: {e}')
    if all_before:
        print(f'All before cutoff, stop')
        break
    page += 1
    if page > 30:
        break
print(f'\nDone. Total: {count} articles imported.')
