#!/usr/bin/env python3
"""
惠城区人民政府 — 建设项目环境影响评价和竣工环保验收
=================================================
CMS: 自建静态HTML列表 + 分页 index_N.html
列表: /hcqzdlyxxgk/hjbhxxgk/jsxmjghbys/index.html → index_2..10.html
详情: /hcqzdlyxxgk/hjbhxxgk/jsxmjghbys/content/post_{N}.html (content in td#zoomcon)
IP: 113.96.111.93 + Host header (no SSL needed)
"""
import sys, os, re, urllib.parse, warnings
warnings.filterwarnings("ignore")

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

import requests
from bs4 import BeautifulSoup

SITE_NAME = "惠城区-建设项目环境影响评价和竣工环保验收"
BASE_DOMAIN = "www.hcq.gov.cn"
IP = "113.96.111.93"
LIST_BASE = f"http://{IP}/hcqzdlyxxgk/hjbhxxgk/jsxmjghbys"
DETAIL_BASE = f"http://{IP}"
TOTAL_PAGES = 10
ITEMS_PER_PAGE = 15

HEADERS = {
    "Host": BASE_DOMAIN,
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
}


def fetch(url, retries=3):
    for attempt in range(retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            r.encoding = 'utf-8'
            return r.text
        except Exception as e:
            if attempt < retries - 1:
                import time
                time.sleep(2)
            else:
                print(f"  [ERROR] {url[:80]}: {e}", file=sys.stderr)
                return None


def fetch_list_page(page_no):
    """Fetch list page, return list of (title, url, date)"""
    if page_no == 1:
        url = f"{LIST_BASE}/index.html"
    else:
        url = f"{LIST_BASE}/index_{page_no}.html"
    text = fetch(url)
    if not text:
        return []
    soup = BeautifulSoup(text, 'html.parser')
    items = []
    for div in soup.select('div[id="fathernodid8"]'):
        a = div.find('a')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get('title', '') or a.get_text(strip=True)
        # Find date in adjacent td
        date_td = div.find('td', width='17%')
        date = date_td.get_text(strip=True) if date_td else ""
        if href and not href.startswith('http'):
            href = urllib.parse.urljoin(DETAIL_BASE, href)
        # Rewrite www.hcq.gov.cn to IP-based URL for direct access
        href = re.sub(r'https?://www\.hcq\.gov\.cn', f'http://{IP}', href)
        items.append({"title": title.strip(), "url": href, "date": date.strip()})
    return items


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_detail(detail_url):
    """Extract content from detail page (td#zoomcon)"""
    text = fetch(detail_url)
    if not text:
        return None, None, None, []
    soup = BeautifulSoup(text, 'html.parser')
    # Title from meta tag (actual article title)
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    title = meta_title.get('content', '').strip() if meta_title else ""
    if not title:
        title_tag = soup.find('title')
        title = title_tag.get_text(strip=True) if title_tag else ""
        title = re.sub(r'_重点领域信息公开$', '', title).strip()
    # Date from meta tag
    date_str = None
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date:
        pub_content = meta_date.get('content', '')
        date_m = re.match(r'(\d{4}-\d{1,2}-\d{1,2})', pub_content)
        if date_m:
            date_str = date_m.group(1)
    zoom = soup.find(id='zoomcon')
    if not zoom:
        return title, date_str, None, []
    # Extract body: iterate children of zoomcon in order
    body_parts = []
    attachments = []
    for el in zoom.find_all(['p', 'table', 'img'], recursive=False):
        if el.name == 'p':
            txt = el.get_text(" ", strip=True)
            if txt:
                body_parts.append(txt)
        elif el.name == 'table':
            md = html_table_to_html(el)
            if md:
                body_parts.append(md)
        elif el.name == 'img':
            src = el.get('src', '')
            alt = el.get('alt', '')
            if src:
                full_src = urllib.parse.urljoin(DETAIL_BASE, src) if not src.startswith('http') else src
                body_parts.append(f"![{alt}]({full_src})")
    # Also check for deeply nested content (p inside div etc.)
    for el in zoom.find_all(['div', 'span'], recursive=False):
        txt = el.get_text(" ", strip=True)
        if txt and len(txt) > 20:
            body_parts.append(txt)
    # Attachments: links to PDF/DOC inside zoomcon
    for a in zoom.find_all('a'):
        href = a.get('href', '')
        onclick = a.get('onclick', '')
        fname = a.get_text(strip=True)
        # Direct PDF/DOC links
        if any(ext in href.lower() for ext in ['.pdf', '.doc', '.xls', '.zip']):
            full_url = re.sub(r'https?://www\.hcq\.gov\.cn', f'http://{IP}', href)
            if not href.startswith('http'):
                full_url = urllib.parse.urljoin(DETAIL_BASE, href)
            attachments.append({"name": fname, "url": full_url})
        # Onclick download
        elif 'pdf' in onclick.lower() or 'download' in onclick.lower():
            url_match = re.search(r"['\"]([^'\"]+\.pdf)['\"]", onclick)
            if url_match:
                full_url = urllib.parse.urljoin(DETAIL_BASE, url_match.group(1))
                attachments.append({"name": fname, "url": full_url})
    body = "\n\n".join(body_parts) if body_parts else None
    return title, date_str, body, attachments


def crawl(pages=5):
    all_items = []
    max_pages = min(pages, TOTAL_PAGES)
    for page_no in range(1, max_pages + 1):
        items = fetch_list_page(page_no)
        if not items:
            break
        all_items.extend(items)
        print(f"Page {page_no}: {len(items)} items", flush=True)
    print(f"\nTotal list items: {len(all_items)}", flush=True)
    results = []
    empty_body = 0
    with_attach = 0
    for i, item in enumerate(all_items):
        title = item["title"]
        detail_url = item["url"]
        list_date = item["date"]
        parsed_title, pub_date, body, attachments = parse_detail(detail_url)
        final_title = parsed_title or title
        final_date = pub_date or list_date
        attach_str = ";".join([a["name"] for a in attachments]) if attachments else ""
        if attachments:
            with_attach += 1
        if not body:
            empty_body += 1
        results.append({
            "site_name": SITE_NAME,
            "title": final_title,
            "url": detail_url,
            "source_url": detail_url,
            "pub_date": final_date,
            "content": body or "",
            "summary": (re.sub(r'\s+', ' ', body or "")[:300]) if body else "",
            "attachments": attach_str,
        })
        if (i + 1) % 10 == 0:
            print(f"  Parsed: {i+1}/{len(all_items)}", flush=True)
    print(f"Parsed: {len(results)} items ({empty_body} empty body, {with_attach} with attachments)", flush=True)
    push_to_searchdb(results, "hcq_hjbh")


if __name__ == "__main__":
    pages = int(sys.argv[sys.argv.index("--pages") + 1]) if "--pages" in sys.argv else 5
    crawl(pages=pages)
