#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""crawl_taihe_hp.py — 太和县生态环境分局-建设项目环评文件审批"""

import os, re, sqlite3, time, sys, random, urllib.parse
from datetime import datetime, timedelta
from concurrent.futures import ThreadPoolExecutor, as_completed

DB_PATH = os.environ.get('SEARCH_DB', '/root/search.db')
SITE_NAME = '太和县生态环境分局-建设项目环评文件审批'
BASE = 'https://www.taihe.gov.cn'
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime('%Y-%m-%d')
MAX_WORKERS = 5

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36',
    'X-Requested-With': 'XMLHttpRequest',
}

import requests
sess = requests.Session()
sess.headers.update(HEADERS)


def fetch(url, timeout=15):
    for _ in range(3):
        try:
            r = sess.get(url, timeout=timeout)
            r.encoding = 'utf-8'
            return r.text
        except Exception as e:
            time.sleep(1)
    return None


def parse_list_html(html):
    """Parse articles from AJAX list HTML"""
    items = []
    ul_match = re.search(r'<ul>(.*?)</ul>', html, re.DOTALL)
    if not ul_match:
        return items
    
    for li in re.finditer(r'<li>(.*?)</li>', ul_match.group(1), re.DOTALL):
        content = li.group(1)
        a = re.search(r'<a\s+href="([^"]+)"\s+title="([^"]*)"', content)
        span = re.search(r'<span>([^<]+)</span>', content)
        if a and span:
            items.append({
                'url': urllib.parse.urljoin(BASE, a.group(1)),
                'title': a.group(2).strip(),
                'date': span.group(1).strip(),
            })
    return items


def parse_detail(html):
    """Extract title and content from detail page"""
    title_m = re.search(r'<meta\s+name="ArticleTitle"\s+content="([^"]*)"', html)
    title = title_m.group(1) if title_m else ''
    
    if not title:
        t = re.search(r'<div class="text-center u-title">(.*?)</div>', html, re.DOTALL)
        title = t.group(1).strip() if t else ''
    
    content = ''
    m = re.search(r'<div class="g-detailbox[^"]*"\s+id="zoom"[^>]*>(.*?)</div>\s*</div>\s*</div>\s*</div>', html, re.DOTALL)
    if m:
        content = m.group(1).strip()
    
    # Clean
    content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL)
    content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL)
    content = re.sub(r'<o:p>\s*</o:p>', '', content)
    content = re.sub(r'<o:p/>', '', content)
    content = re.sub(r'<p[^>]*>\s*(?:<br\s*/?>\s*)*</p>', '', content)
    
    return title, content.strip()


def process_item(item):
    """Fetch detail and return insert data"""
    html = fetch(item['url'])
    if not html:
        return None, f'FETCH_ERR: {item["title"][:30]}'
    
    dt, content = parse_detail(html)
    final_title = dt or item['title']
    
    return {
        'site_name': SITE_NAME,
        'title': final_title,
        'page_url': item['url'],
        'publish_date': item['date'],
        'source_url': item['url'],
        'content': content,
    }, None


def main():
    print(f'[{datetime.now().strftime("%H:%M:%S")}] {SITE_NAME}')
    
    # Step 1: Get first page to determine total
    html = fetch('https://www.taihe.gov.cn/OpennessTarget/350/29374/page_1.html')
    if not html:
        print('[ERROR] Failed to fetch page 1')
        return
    
    # Extract page count
    pc = re.search(r'pagecount="(\d+)"', html)
    total_pages = int(pc.group(1)) if pc else 39
    print(f'Total pages: {total_pages}')
    
    # Step 2: Fetch all list pages
    all_items = parse_list_html(html)
    for p in range(2, total_pages + 1):
        h = fetch(f'https://www.taihe.gov.cn/OpennessTarget/350/29374/page_{p}.html')
        if h:
            items = parse_list_html(h)
            all_items.extend(items)
        else:
            print(f'  [WARN] Page {p} failed')
        time.sleep(random.uniform(0.2, 0.4))
    
    # Dedup by URL
    seen = set()
    unique = []
    for item in all_items:
        if item['url'] not in seen:
            seen.add(item['url'])
            unique.append(item)
    
    print(f'Total unique: {len(unique)}')
    
    # Filter by 3-year cutoff
    recent = [i for i in unique if i['date'] >= CUTOFF]
    print(f'Within 3 years: {len(recent)} (filtered: {len(unique)-len(recent)})')
    
    # Step 3: Check existing in DB
    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    existing = set()
    for row in c.execute('SELECT page_url FROM gov_raw WHERE site_name=?', (SITE_NAME,)):
        existing.add(row[0])
    
    to_fetch = [i for i in recent if i['url'] not in existing]
    print(f'New items: {len(to_fetch)}')
    
    # Step 4: Multi-threaded detail fetch
    new_count = error_count = 0
    
    with ThreadPoolExecutor(max_workers=MAX_WORKERS) as executor:
        futures = {executor.submit(process_item, item): item for item in to_fetch}
        for i, future in enumerate(as_completed(futures), 1):
            item = futures[future]
            result, err = future.result()
            
            if err or not result:
                print(f'  [{i}/{len(to_fetch)}] {item["title"][:30]}... {err or "FAIL"}')
                error_count += 1
                continue
            
            try:
                c.execute('''INSERT OR IGNORE INTO gov_raw
                    (site_name, title, page_url, publish_date, source_url, content)
                    VALUES (?,?,?,?,?,?)''',
                    (result['site_name'], result['title'], result['page_url'],
                     result['publish_date'], result['source_url'], result['content']))
                if c.rowcount:
                    new_count += 1
                    status = f'OK ({len(result["content"])}B)'
                else:
                    status = 'DUP'
                print(f'  [{i}/{len(to_fetch)}] {item["title"][:30]}... {status}')
            except Exception as e:
                print(f'  [{i}/{len(to_fetch)}] {item["title"][:30]}... ERR: {e}')
                error_count += 1
            
            conn.commit()
    
    conn.close()
    print(f'\n=== Done === New: {new_count}, Errors: {error_count}')


if __name__ == '__main__':
    main()
