#!/usr/bin/env python3
"""
Crawl 广东环科技术咨询有限公司 - 公示公告
https://www.gdhuanke.com/category/notice
Pagination: ?page=N&limit=15 (0-indexed)
List: <dl><dt>日期</dt><dd><div class="title"><a href="..." title="完整标题">标题</a></div></dd></dl>
Detail: <h2>标题</h2> + <div class="article_bline">日期</div> + <div class="article_content">HTML</div>
"""

import requests, json, re, os, sys, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = 'https://www.gdhuanke.com/category/notice'
OUTPUT_FILE = '/root/gov_crawler/output/gdhuanke_notice.jsonl'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36',
    'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
}

os.makedirs(os.path.dirname(OUTPUT_FILE), exist_ok=True)

session = requests.Session()
session.headers.update(HEADERS)

def fetch_list_page(page_num):
    url = f'{BASE_URL}?page={page_num}&limit=15'
    resp = session.get(url, timeout=30)
    resp.encoding = 'utf-8'
    soup = BeautifulSoup(resp.text, 'html.parser')
    items = []
    for dl in soup.find_all('dl', class_='clearfix'):
        dt = dl.find('dt')
        dd = dl.find('dd')
        date = dt.get_text(strip=True) if dt else ''
        if dd:
            a = dd.find('a')
            if a and a.get('href'):
                href = a['href']
                title = a.get('title', '') or a.get_text(strip=True)
                if not href.startswith('http'):
                    href = urljoin(BASE_URL, href)
                items.append({'url': href, 'title': title.strip(), 'date': date})
    return items

def parse_detail(url, html):
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from h2
    h2 = soup.find('h2')
    title = h2.get_text(strip=True) if h2 else ''
    
    # Date from article_bline
    date = ''
    bline = soup.find('div', class_='article_bline')
    if bline:
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', bline.get_text())
        if m:
            date = m.group(1)
    
    # Content from article_content
    content_div = soup.find('div', class_='article_content')
    content_html = ''
    summary = ''
    attachments = []
    if content_div:
        # Keep structure but clean inline styles
        for tag in content_div.find_all(True):
            keep = ['href', 'src', 'alt', 'target']
            for attr in list(tag.attrs):
                if attr not in keep:
                    del tag[attr]
        content_html = str(content_div)
        summary = content_div.get_text('\n', strip=True)
        
        # Attachments
        for a in content_div.find_all('a', href=True):
            href = a['href']
            if any(href.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                full_url = href if href.startswith('http') else urljoin(url, href)
                attachments.append({'url': full_url, 'text': a.get_text(strip=True) or os.path.basename(href)})
    
    return {
        'title': title,
        'pub_date': date,
        'content_html': content_html,
        'summary': summary[:500] if summary else '',
        'attachments': attachments
    }

def crawl_all():
    all_items = []
    seen_titles = set()
    page = 0
    empty_page_count = 0
    
    # Keep fetching until we hit 3 empty pages in a row (pagination loop detection)
    while empty_page_count < 3:
        print(f'  Page {page}...')
        try:
            items = fetch_list_page(page)
            if not items:
                empty_page_count += 1
                print(f'    Empty page (count: {empty_page_count})')
                page += 1
                time.sleep(0.1)
                continue
            
            empty_page_count = 0
            new_count = 0
            for item in items:
                title = item['title']
                if title not in seen_titles:
                    seen_titles.add(title)
                    all_items.append(item)
                    new_count += 1
            
            print(f'    {len(items)} items, {new_count} new')
            
            # If no new items, we've wrapped around (need 5+ pages of duplicates)
            if new_count == 0 and page >= 5:
                print('    No new items - stopping (wrapped around)')
                break
                
        except Exception as e:
            print(f'    [ERROR] {e}')
            empty_page_count += 1
        
        page += 1
        time.sleep(0.3)
    
    print(f'\nTotal unique items: {len(all_items)}')
    
    # Crawl details
    record_count = 0
    with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:
        for i, item in enumerate(all_items):
            print(f'  [{i+1}/{len(all_items)}] {item["title"][:40]}...')
            try:
                resp = session.get(item['url'], timeout=30)
                resp.encoding = 'utf-8'
                detail = parse_detail(item['url'], resp.text)
                
                title = detail['title'] or item['title']
                
                record = {
                    'title': title,
                    'url': item['url'],
                    'date': detail['pub_date'] or item['date'],
                    'content': detail['content_html'],
                    'summary': detail['summary'],
                    'site_name': '广东环科技术咨询有限公司',
                    'group': '企业',
                    'attachments': json.dumps(detail['attachments'], ensure_ascii=False) if detail['attachments'] else '',
                }
                f.write(json.dumps(record, ensure_ascii=False) + '\n')
                record_count += 1
            except Exception as e:
                print(f'    [ERROR] {item["url"]}: {e}')
            time.sleep(0.2)
    
    print(f'\nDone! {record_count} records written to {OUTPUT_FILE}')

if __name__ == '__main__':
    if '--incremental' in sys.argv:
        print('Incremental mode')
        items = fetch_list_page(0)
        print(f'Found {len(items)} items on page 1')
        with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:
            for item in items:
                try:
                    resp = session.get(item['url'], timeout=30)
                    resp.encoding = 'utf-8'
                    detail = parse_detail(item['url'], resp.text)
                    title = detail['title'] or item['title']
                    record = {
                        'title': title,
                        'url': item['url'],
                        'date': detail['pub_date'] or item['date'],
                        'content': detail['content_html'],
                        'summary': detail['summary'],
                        'site_name': '广东环科技术咨询有限公司',
                        'group': '企业',
                        'attachments': json.dumps(detail['attachments'], ensure_ascii=False) if detail['attachments'] else '',
                    }
                    f.write(json.dumps(record, ensure_ascii=False) + '\n')
                except Exception as e:
                    print(f'ERROR: {e}')
                time.sleep(0.2)
        print(f'Incremental done: {len(items)} records')
    else:
        crawl_all()
