#!/usr/bin/env python3
"""
宁国市基层政务公开 - 生态环境 爬虫
https://www.ningguo.gov.cn/Jczwgk/showList/0/111000000/page_1.html
"""
import requests
from bs4 import BeautifulSoup
import re, os, sys, time
from urllib.parse import urljoin

BASE_URL = "https://www.ningguo.gov.cn"
LIST_BASE = "https://www.ningguo.gov.cn/Jczwgk/showList/0/111000000/page"
SITE_NAME = "宁国市生态环境"
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=15)
        import chardet
        r.encoding = chardet.detect(r.content)['encoding'] or 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERROR] fetch failed: {url} - {e}")
        return None

def extract_list_items(html):
    """Extract (url, title, date) from list page."""
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for a in soup.find_all('a', href=True):
        href = a['href']
        if '/Jczwgk/show/' in href and href.endswith('.html'):
            full_url = urljoin(BASE_URL, href)
            # Get parent li for date
            li = a.find_parent('li')
            li_text = li.get_text(strip=True) if li else a.get_text(strip=True)
            title = a.get_text(strip=True)
            # Extract date from li text prefix: "YYYY-MM-DDTitle"
            date_match = re.match(r'(\d{4}-\d{2}-\d{2})\s*', li_text)
            date = date_match.group(1) if date_match else ''
            if title:
                items.append((full_url, title, date))
    return items

def extract_detail(html, page_url):
    """Extract title, content, attachments from detail page."""
    soup = BeautifulSoup(html, 'html.parser')
    
    # 1. Title from <h1 class="u-lgtit"> or <title>
    title_el = soup.find('h1', class_='u-lgtit')
    if title_el:
        title = title_el.get_text(strip=True)
    else:
        title_tag = soup.find('title')
        title = title_tag.get_text(strip=True) if title_tag else ""
        title = re.sub(r'\s*[-–—]\s*标准化规范化工作专题\s*$', '', title)
        title = re.sub(r'\s*[-–—]\s*宁国市人民政府\s*$', '', title)
    
    # 2. Date from u-wzinfo
    date = ""
    wzinfo = soup.find('div', class_='u-wzinfo')
    if wzinfo:
        date_match = re.search(r'(\d{4})-(\d{2})-(\d{2})', wzinfo.get_text())
        if date_match:
            date = f"{date_match.group(1)}-{date_match.group(2)}-{date_match.group(3)}"
    
    # 3. Content + Attachments
    content_parts = []
    attachments = []
    
    zoom = soup.find('div', class_='j-fontContent')
    if zoom:
        # Attachments
        for a in zoom.find_all('a', href=True):
            href = a['href']
            if any(kw in href.lower() for kw in ['download', '.pdf', '.doc', '.docx', '.xls', '.xlsx', '.rar', '.zip']):
                full_url = urljoin(page_url, href)
                attach_name = a.get_text(strip=True) or href.split('/')[-1]
                if attach_name not in ['文本下载']:  # skip generic "文本下载" link
                    attachments.append({'name': attach_name, 'url': full_url})
                elif href.startswith('/upload_xc/') or 'download' in href:
                    # Generic download links that point to actual files
                    attachments.append({'name': attach_name, 'url': full_url})
        
        # Content: paragraphs, tables, images
        title_skip = title[:80]
        for el in zoom.find_all(['p', 'table', 'img']):
            if el.name == 'p':
                if el.find_parent('table'):
                    continue
                txt = el.get_text(strip=True)
                if txt:
                    if title_skip and txt == title:
                        continue
                    content_parts.append(txt)
            elif el.name == 'table':
                tbl_html = html_table_to_html(el, page_url)
                if tbl_html:
                    content_parts.append(tbl_html)
            elif el.name == 'img':
                src = el.get('src', '')
                alt = el.get('alt', '')
                if src:
                    content_parts.append(f'![{alt}]({urljoin(page_url, src)})')
    
    content = '\n\n'.join(content_parts)
    
    # Fallback: if no p/table/img produced content but there's text in zoom
    if not content and zoom:
        raw = zoom.get_text(strip=True)
        if raw:
            content = raw
            # Also check for attachments in fallback mode
            for a in zoom.find_all('a', href=True):
                href = a['href']
                if any(kw in href.lower() for kw in ['download', '.pdf', '.doc', '.docx', '.xls']):
                    full_url = urljoin(page_url, href)
                    attach_name = a.get_text(strip=True) or href.split('/')[-1]
                    if attach_name not in ['文本下载']:
                        attachments.append({'name': attach_name, 'url': full_url})
                    elif href.startswith('/upload_xc/') or 'download' in href:
                        attachments.append({'name': attach_name, 'url': full_url})
    
    return title, content, attachments, date

def main():
    items_all = []
    
    # Step 1: Collect items from pages 1-5
    for page in range(1, MAX_PAGES + 1):
        url = f"{LIST_BASE}_{page}.html"
        print(f"Fetching list page {page}: {url}")
        html = fetch(url)
        if not html:
            print(f"  [SKIP] Page {page} failed")
            continue
        
        items = extract_list_items(html)
        print(f"  Found {len(items)} items")
        items_all.extend(items)
        time.sleep(0.3)
    
    print(f"\nTotal items to process: {len(items_all)}")
    
    # Step 2: Fetch each detail page
    results = []
    for i, (page_url, list_title, list_date) in enumerate(items_all):
        print(f"[{i+1}/{len(items_all)}] {list_title[:50]}...")
        
        html = fetch(page_url)
        if not html:
            results.append({
                'title': list_title, 'content': '', 'page_url': page_url,
                'attachments': [], 'date': list_date,
            })
            continue
        
        title, content, attachments, date = extract_detail(html, page_url)
        if not title:
            title = list_title
        if not date:
            date = list_date
        
        results.append({
            'title': title, 'content': content, 'page_url': page_url,
            'attachments': attachments, 'date': date,
        })
        
        print(f"  -> {title[:50]} | date={date} | attach={len(attachments)} | content_len={len(content)}")
        time.sleep(0.2)
    
    # Step 3: Push to search DB
    SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
    
    try:
        import sqlite3
        conn = sqlite3.connect(SEARCH_DB, timeout=60)
        c = conn.cursor()
        
        inserted = 0
        for r in results:
            summary = r['content'][:200] if r['content'] else ''
            attachments_json = str(r['attachments']) if r['attachments'] else ''
            has_table = 1 if '| --- |' in r['content'] else 0
            c.execute(
                """INSERT OR REPLACE INTO gov_raw (title, content, page_url, source_url, site_name, publish_date, summary, attachments, has_table, category, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, '环评审批', 'crawl_ningguo_hjbh.py')""",
                (r['title'], r['content'], r['page_url'], r['page_url'], SITE_NAME,
                 r['date'], summary, attachments_json, has_table)
            )
            inserted += 1
        
        conn.commit()
        conn.close()
        print(f"\n✅ Pushed {inserted} records to search DB")
        
    except Exception as e:
        print(f"\n[ERROR] DB push failed: {e}")
    
    print(f"\n{'='*50}")
    print(f"站点: {SITE_NAME}")
    print(f"采集页数: {MAX_PAGES}")
    print(f"采集条数: {len(results)}")
    print(f"{'='*50}")

if __name__ == '__main__':
    main()
