#!/usr/bin/env python3
"""
Crawler for 清徐经济开发区 - 通知公告
Local crawl + JSON output for server insert
"""

import re, sys, time, json
from urllib.request import urlopen, Request
from urllib.parse import urljoin

BASE_URL = 'https://www.qxedz.com'
SITE_NAME = '清徐经济开发区-通知公告'
MAX_PAGES = 5
HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}


def fetch(url):
    req = Request(url, headers=HEADERS)
    resp = urlopen(req, timeout=30)
    return resp.read().decode('utf-8')


def main():
    start_time = time.time()
    
    # Build all list page URLs
    page_urls = [f'{BASE_URL}/tzgg.html']
    for p in range(2, MAX_PAGES + 1):
        page_urls.append(f'{BASE_URL}/tzgg/list_94_{p}.html')
    
    # Collect all article URLs from list pages
    all_articles = []
    for page_url in page_urls:
        print(f'[qxedz] Fetching list: {page_url}', file=sys.stderr)
        try:
            html = fetch(page_url)
            # Parse articles from list page
            pattern = r'<a[^>]*href=[\"\']([^\"\']+?\.html)[\"\'][^>]*title=[\"\']([^\"\']*)[\"\']>\s*<span>(.*?)</span>\s*<font>(.*?)</font>\s*</a>'
            found = 0
            for m in re.finditer(pattern, html, re.DOTALL):
                url_path = m.group(1).strip()
                if url_path.startswith('/tzgg/') and url_path.count('/') == 2:
                    title = m.group(2).strip() or m.group(3).strip()
                    date = m.group(4).strip()
                    full_url = BASE_URL + url_path
                    all_articles.append({'url': full_url, 'title': title, 'date': date})
                    found += 1
            print(f'[qxedz]   Found {found} articles (total: {len(all_articles)})', file=sys.stderr)
            time.sleep(0.3)
        except Exception as e:
            print(f'[qxedz]   Error: {e}', file=sys.stderr)
            break
    
    print(f'[qxedz] Crawling {len(all_articles)} detail pages...', file=sys.stderr)
    
    # Crawl each detail page
    results = []
    for i, art in enumerate(all_articles):
        try:
            html = fetch(art['url'])
            
            # Title from detail page
            t_match = re.search(r'<div class="newsxxtitle">\s*<span>(.*?)</span>', html, re.DOTALL)
            title = t_match.group(1).strip() if t_match else art['title']
            
            # Date from detail page
            d_match = re.search(r'发布时间[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})', html)
            date = d_match.group(1).strip() if d_match else art['date']
            
            # Content from InfoContent
            c_match = re.search(r'<div[^>]*class=[\"\']?InfoContent[\"\']?[^>]*>(.*?)</div>\s*<div[^>]*class=[\"\']?(?:clear|page)[\"\']?', html, re.DOTALL)
            if not c_match:
                c_match = re.search(r'<div[^>]*class=[\"\']?InfoContent[\"\']?[^>]*>(.*?)</div>', html, re.DOTALL)
            
            content = ''
            if c_match:
                content = c_match.group(1).strip()
                # Clean scripts/styles
                content = re.sub(r'<script[^>]*>.*?</script>', '', content, flags=re.DOTALL | re.I)
                content = re.sub(r'<style[^>]*>.*?</style>', '', content, flags=re.DOTALL | re.I)
                content = content.strip()
            
            results.append({
                'title': title,
                'content': content,
                'date': date[:10],
                'url': art['url'],
            })
            
            if (i + 1) % 10 == 0:
                print(f'[qxedz] Progress: {i+1}/{len(all_articles)}', file=sys.stderr)
            
            time.sleep(0.2)
        except Exception as e:
            print(f'[qxedz] Error on {art["url"]}: {e}', file=sys.stderr)
    
    # Output JSON for server import
    print(json.dumps({
        'site_name': SITE_NAME,
        'articles': results,
        'stats': {
            'total': len(results),
            'with_content': sum(1 for r in results if r['content'] and len(r['content']) > 50),
            'elapsed': time.time() - start_time,
        }
    }, ensure_ascii=False))
    
    print(f'[qxedz] Done: {len(results)} articles in {time.time()-start_time:.1f}s', file=sys.stderr)


if __name__ == '__main__':
    main()
