#!/usr/bin/env python3
"""
广丰区生态环境局 — 环评信息
================================
CMS: UCAP CMS
列表: /qsthjj/HPXXS/zwgk_list.shtml (第1页), zwgk_list_{page}.shtml (第2页起)
详情: /qsthjj/HPXXS/{year}/{id}.shtml (div#zoomcon > ucapcontent > p)
"""
import sys, os, re, urllib.parse, warnings
warnings.filterwarnings("ignore")

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

import requests
from bs4 import BeautifulSoup

SITE_NAME = "广丰区生态环境局-环评信息"
BASE_URL = "https://www.gfx.gov.cn"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
}

PAGE_SIZE = 16


def fetch_page(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        return None


def fetch_list_page(page_no):
    """Fetch list page: base URL for page 1, zwgk_list_{page}.shtml for page 2+"""
    if page_no == 1:
        url = f"{BASE_URL}/qsthjj/HPXXS/zwgk_list.shtml"
    else:
        url = f"{BASE_URL}/qsthjj/HPXXS/zwgk_list_{page_no}.shtml"
    
    html = fetch_page(url)
    if not html:
        return [], False
    
    soup = BeautifulSoup(html, 'html.parser')
    items = soup.select('ul#doclist > li')
    
    results = []
    for li in items:
        a = li.find('a')
        span = li.find('span')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get('title', '')  # Full title from @title attribute
        if not title:
            title = a.get_text(strip=True)  # Fallback to truncated text
        date_str = span.get_text(strip=True) if span else ''
        if href and not href.startswith('http'):
            href = urllib.parse.urljoin(BASE_URL, href)
        results.append({"title": title, "url": href, "date": date_str})
    
    has_more = len(items) >= PAGE_SIZE
    return results, has_more


def parse_detail(detail_url):
    """Extract content from detail page (div#zoomcon > ucapcontent > p)"""
    html = fetch_page(detail_url)
    if not html:
        return None, None, None, []
    
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from <title> — this site only has site name here, not article title
    # We return None so the crawler uses the list page's @title attribute
    
    # Date from page (look for publish time patterns)
    date_str = None
    date_m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})', html)
    if date_m:
        date_str = date_m.group(1)
    
    # Content container
    zoomcon = soup.find('div', id='zoomcon') or soup.find('div', class_='zwcon')
    if not zoomcon:
        return None, date_str, None, []
    
    ucap = zoomcon.find('ucapcontent')
    content_root = ucap if ucap else zoomcon
    
    body_parts = []
    attachments = []
    
    # Extract all <p> elements
    for p in content_root.find_all('p'):
        # Check for <a> with attachments
        for a in p.find_all('a'):
            href = a.get('href', '')
            fname = a.get_text(strip=True)
            if href and re.search(r'\.(pdf|doc|docx|xls|xlsx|rar|zip|ppt|pptx)$', href, re.I):
                full_url = urllib.parse.urljoin(BASE_URL, href)
                attachments.append({"name": fname, "url": full_url})
                if fname:
                    body_parts.append(f"- [{fname}]({full_url})")
        
        # Get paragraph text with <br> → newline handling
        DELIM = '\x00BR\x00'
        for br in p.find_all('br'):
            br.replace_with(DELIM)
        txt = p.get_text(" ", strip=True)
        txt = txt.replace(DELIM, '\n')
        txt = re.sub(r'\n\s*\n', '\n\n', txt)
        
        if txt and len(txt) > 5:
            body_parts.append(txt)
    
    # Also check for tables
    for table in content_root.find_all('table'):
        md = html_table_to_html(table, base_url=BASE_URL)
        if md:
            body_parts.append(md)
    
    body = "\n\n".join(body_parts) if body_parts else None
    return None, date_str, body, attachments


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def crawl(pages=5):
    all_items = []
    for page_no in range(1, pages + 1):
        items, has_more = fetch_list_page(page_no)
        if not items:
            break
        all_items.extend(items)
        print(f"Page {page_no}: {len(items)} items", flush=True)
        if not has_more:
            break
    
    print(f"\nTotal list items: {len(all_items)}", flush=True)
    
    results = []
    empty_body = 0
    with_attach = 0
    
    for i, item in enumerate(all_items):
        title = item["title"]
        detail_url = item["url"]
        list_date = item["date"]
        parsed_title, pub_date, body, attachments = parse_detail(detail_url)
        final_title = parsed_title or title
        final_date = pub_date or list_date
        attach_str = ";".join([f"[{a['name']}]({a['url']})" for a in attachments]) if attachments else ""
        if attachments:
            with_attach += 1
        if not body:
            empty_body += 1
        results.append({
            "site_name": SITE_NAME,
            "title": final_title,
            "url": detail_url,
            "source_url": detail_url,
            "pub_date": final_date,
            "content": body or "",
            "summary": (re.sub(r'\s+', ' ', body or "")[:300]) if body else "",
            "attachments": attach_str,
        })
        if (i + 1) % 10 == 0:
            print(f"  Parsed: {i+1}/{len(all_items)}", flush=True)
    
    print(f"Parsed: {len(results)} items ({empty_body} empty body, {with_attach} with attachments)", flush=True)
    push_to_searchdb(results, "guangfeng_hpxx")


if __name__ == "__main__":
    pages = int(sys.argv[sys.argv.index("--pages") + 1]) if "--pages" in sys.argv else 5
    crawl(pages=pages)
