#!/usr/bin/env python3
"""射洪-生态环境 爬虫"""
import os, re, sqlite3, time
from datetime import datetime, timezone, timedelta
from urllib.parse import urljoin



import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
BASE = 'https://www.shehong.gov.cn'
SITE_NAME = '射洪生态环境'
DB = os.getenv("SEARCH_DB", "/root/search.db")
LIST_URL = BASE + '/gongkai/kuozhan/10439.html?page={page}'
UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36'

tz = timezone(timedelta(hours=8))
cutoff = datetime.now(tz) - timedelta(days=365*3)
print(f'Cutoff: {cutoff.isoformat()}')

def extract_xqing(html):
    """Extract content from xqing-web-box div. Search for the quoted version to avoid CSS matches."""
    marker = '"xqing-web-box"'
    idx = html.find(marker)
    if idx < 0:
        marker = "'xqing-web-box'"
        idx = html.find(marker)
    if idx < 0:
        return ''
    
    div_start = html.rfind('<div', 0, idx)
    if div_start < 0:
        return ''
    
    open_gt = html.find('>', div_start)
    if open_gt < 0:
        return ''
    
    start = open_gt + 1
    depth = 1
    i = start
    while depth > 0 and i < len(html):
        if html[i:i+4] == '<div':
            skip = html.find('>', i)
            if skip > i and skip - i < 100:
                tag = html[i+1:skip]
                if not tag.startswith('/') and not tag.startswith('!--') and not tag.startswith('['):
                    depth += 1
                i = skip + 1
            else:
                i += 1
        elif html[i:i+5] == '</div':
            close_gt = html.find('>', i)
            if close_gt > 0:
                depth -= 1
                i = close_gt + 1
            else:
                i += 1
        else:
            i += 1
    
    if depth == 0:
        return html[start:i-6].strip()
    return ''

count = 0
page = 1
while True:
    url = LIST_URL.format(page=page)
    html = os.popen(f'curl -s --max-time 15 -H "User-Agent: {UA}" "{url}"').read()
    
    if not html.strip() or '被拦截' in html:
        print(f'Page {page}: blocked, waiting 30s...')
        time.sleep(30)
        html = os.popen(f'curl -s --max-time 15 -H "User-Agent: {UA}" "{url}"').read()
        if not html.strip() or '被拦截' in html:
            print(f'Page {page}: still blocked, stop')
            break
    
    m = re.search(r'<ul class="content-list"[^>]*>(.*?)</ul>', html, re.DOTALL)
    if not m:
        print(f'Page {page}: no list, stop')
        break
    
    items = m.group(1)
    lis = re.findall(r'<li[^>]*>(.*?)</li>', items, re.DOTALL)
    if not lis:
        print(f'Page {page}: no items, stop')
        break
    
    print(f'Page {page}: {len(lis)} items')
    all_before = True
    
    for li in lis:
        m_date = re.search(r'(\d{4}-\d{2}-\d{2})', li)
        list_date = m_date.group(1) if m_date else ''
        
        if list_date < cutoff.strftime('%Y-%m-%d'):
            continue
        all_before = False
        
        m_link = re.search(r'href=["\']([^"\']+)["\']', li)
        if not m_link:
            continue
        link = m_link.group(1)
        if not link.startswith('http'):
            link = urljoin(BASE, link)
        
        time.sleep(1.0)
        
        detail_html = os.popen(f'curl -sL --max-time 15 -H "User-Agent: {UA}" "{link}"').read()
        if not detail_html or len(detail_html) < 500:
            continue
        
        title = ''
        m_t = re.search(r'<title>(.*?)</title>', detail_html, re.DOTALL)
        if m_t:
            title = re.sub(r'\s*-\s*射洪市人民政府\s*', '', m_t.group(1)).strip()
        
        pub_date = list_date
        
        content_html = extract_xqing(detail_html)
        if not content_html or len(content_html) < 20:
            continue
        
        content_html = re.sub(r'<script[^>]*>.*?</script>', '', content_html, flags=re.DOTALL)
        content_html = re.sub(r'<style[^>]*>.*?</style>', '', content_html, flags=re.DOTALL)
        content_html = content_html.strip()
        if not content_html or len(content_html) < 20:
            continue
        
        try:
            conn = sqlite3.connect(DB, timeout=60)
            c = conn.cursor()
            c.execute('INSERT OR REPLACE INTO gov_raw (id, title, content, publish_date, source_url, page_url, site_name, summary) VALUES (?,?,?,?,?,?,?,?)',
                      (None, title, content_html, pub_date, link, link, SITE_NAME, title[:200]))
            conn.commit()
            conn.close()
            count += 1
        except Exception as e:
            print(f'  DB error: {e}')
    
    if all_before:
        print(f'All before cutoff, stop')
        break
    page += 1
    time.sleep(5)
    if page > (_MAX_PG or 60):
        break

print(f'\nDone. Total: {count} articles imported.')
