#!/usr/bin/env python3
"""
yanzhou.gov.cn (兖州区人民政府) — 建设项目环境影响评价爬虫
CMS: Hanweb (汉华) xxgk信息公开系统
列表: POST /module/xxgk/search.jsp (逐页POST)
详情: /art/{year}/{month}/{day}/art_{colId}_{artId}.html
"""

import requests, re, sys, os, time
from datetime import datetime

BASE = "http://www.yanzhou.gov.cn"
SEARCH_URL = f"{BASE}/module/xxgk/search.jsp?standardXxgk=1&infotypeId=YZQA323602&vc_title=&vc_number=&area="
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": f"{BASE}/col/col29302/index.html?vc_xxgkarea=jnsyzq&number=YZQA323602&jh=263",
    "X-Requested-With": "XMLHttpRequest",
    "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8",
}

# Session with cookie persistence
SESS = requests.Session()
SESS.headers.update(HEADERS)

# First visit the page to get session cookie
def init_session():
    try:
        SESS.get(f"{BASE}/col/col29302/index.html?vc_xxgkarea=jnsyzq&number=YZQA323602&jh=263", timeout=15)
    except:
        pass

# 列表POST数据模板
LIST_DATA_TPL = (
    "infotypeId=YZQA323602&jdid=110&divid=div4"
    "&vc_title=&vc_number=&currpage={page}"
    "&standardXxgk=1&area=jnsyzq"
)

CRAWLER_DIR = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, CRAWLER_DIR)
from crawler_lib import push_to_searchdb

MAX_PAGES = int(sys.argv[sys.argv.index("--pages") + 1]) if "--pages" in sys.argv else 5

def fetch_list(page=1):
    """Fetch one page of the EIA list"""
    try:
        r = SESS.post(SEARCH_URL, data=LIST_DATA_TPL.format(page=page), headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
        
        # Extract items: <a title="..." target="..." href="...">
        items = re.findall(
            r'<a\s+title=[\"\x27]([^\"\x27]+)[\"\x27][^>]*href=[\"\x27](https?://[^\"\x27]+?art[^\"\x27]+?)[\"\x27]',
            html
        )
        
        # Extract total pages
        total_pages = 1
        # Try with &nbsp; entities first
        m = re.search(r'共\s*(\d+)\s*页', html.replace('&nbsp;', ' '))
        if m:
            total_pages = int(m.group(1))
        
        # Also extract dates from the items section
        # Items typically show like "7月20日拟作出的建设..." with no explicit date in list
        # We'll get dates from detail pages
        
        return items, total_pages
    except Exception as e:
        print(f"[ERROR] fetch_list page {page}: {e}")
        return [], 0

def fetch_detail(url):
    """Fetch article detail page and extract title, date, content"""
    try:
        r = SESS.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        html = r.text
        
        # Title: <div class="main_tit">
        title = ""
        m = re.search(r'<div class="main_tit">\s*(.*?)\s*</div>', html, re.DOTALL)
        if m:
            raw = m.group(1)
            # Remove HTML comments first (before any tag stripping)
            raw = re.sub(r'<!--.*?-->', '', raw, flags=re.DOTALL)
            # Remove all remaining HTML tags
            title = re.sub(r'<[^>]+>', '', raw).strip()
            # Clean any leftover $[...] template markers
            title = re.sub(r'<\$[^>]+>', '', title).strip()
        
        if not title:
            m = re.search(r'<title>(.*?)</title>', html)
            if m:
                t = m.group(1).replace('兖州区人民政府 建设项目环境影响评价 ', '').strip()
                title = t if t and len(t) > 5 else ""
        
        # Date: <div class="main_riqi"> <span>发布日期：
        publish_date = ""
        m = re.search(r'发布日期：\s*(\d{4}-\d{2}-\d{2})', html)
        if m:
            publish_date = m.group(1)
        
        if not publish_date:
            m = re.search(r"<meta[^>]+Maketime[^>]+content=['\"](\d{4}-\d{2}-\d{2})", html)
            if m:
                publish_date = m.group(1)
        if not publish_date:
            m = re.search(r'成文日期[^>]*>([^<]*\d{4}-\d{2}-\d{2})', html)
            if m:
                publish_date = m.group(1).strip()[-10:]
        
        # Content: <div class="neirong"> ... </div>
        content = ""
        m = re.search(r'<div class="neirong">(.*?)</div>\s*<div class="main-fl-bjxx"', html, re.DOTALL)
        if m:
            raw = m.group(1)
            # Remove meta tags and comments
            raw = re.sub(r'<meta[^>]*>', '', raw)
            raw = re.sub(r'<!--.*?-->', '', raw, flags=re.DOTALL)
            
            # Preserve tables: convert <table> to text format before stripping
            def table_to_text(tbl):
                rows = re.findall(r'<tr[^>]*>.*?</tr>', tbl, re.DOTALL)
                lines = []
                for row in rows:
                    cells = re.findall(r'<t[dh][^>]*>(.*?)</t[dh]>', row, re.DOTALL)
                    # Clean each cell text
                    cell_texts = []
                    for c in cells:
                        ct = re.sub(r'<[^>]+>', '', c)
                        ct = ct.replace('&nbsp;', ' ').strip()
                        cell_texts.append(ct)
                    lines.append(' | '.join(cell_texts))
                # Add Markdown table separator after header if there are at least 2 rows
                if len(lines) >= 2:
                    header_cells = lines[0].split(' | ')
                    sep = ' | '.join(['---'] * len(header_cells))
                    lines.insert(1, sep)
                return '\n'.join(lines)
            
            raw = re.sub(r'<table[^>]*>.*?</table>', lambda m: '\n' + table_to_text(m.group()) + '\n', raw, flags=re.DOTALL)
            
            # Replace </p><p> with double newline for paragraph breaks
            raw = re.sub(r'</p>\s*<p[^>]*>', '\n\n', raw)
            # Replace <br> with newline
            raw = re.sub(r'<br\s*/?>', '\n', raw)
            # Strip remaining HTML tags
            content = re.sub(r'<[^>]+>', '', raw)
            # Decode entities
            content = content.replace('&nbsp;', ' ').replace('&amp;', '&').replace('&lt;', '<').replace('&gt;', '>')
            # Normalize whitespace
            content = re.sub(r'[ \t]+', ' ', content)
            content = re.sub(r'\n{3,}', '\n\n', content)
            content = content.strip()
        
        # Source
        source = ""
        m = re.search(r'信息来源[：:]<[^>]*>[^<]*begin-->(.*?)<!', html)
        if m:
            source = m.group(1).strip()
        m = re.search(r'发布机构[^>]*>([^<]+)', html)
        if m and not source:
            source = m.group(1).strip()
        
        return {
            "title": title,
            "date": publish_date,
            "content": content,
            "source": source,
        }
    except Exception as e:
        print(f"[ERROR] fetch_detail: {e}")
        return None

def main():
    init_session()
    all_items = []
    total_pages = 1
    
    # Fetch page 1 to get total pages
    items, total_pages = fetch_list(1)
    if not items:
        print("[ERROR] No items fetched")
        sys.exit(1)
    
    all_items.extend(items)
    print(f"[INFO] Page 1: {len(items)} items, total pages: {total_pages}")
    
    # Fetch remaining pages up to MAX_PAGES
    pages_to_fetch = min(total_pages, MAX_PAGES)
    for page in range(2, pages_to_fetch + 1):
        items, _ = fetch_list(page)
        if items:
            all_items.extend(items)
            print(f"[INFO] Page {page}: {len(items)} items")
        time.sleep(0.5)
    
    print(f"[INFO] Total items: {len(all_items)}")
    
    batch = []
    for idx, (title_from_list, url) in enumerate(all_items):
        detail = fetch_detail(url)
        if detail is None:
            continue
        
        title = detail["title"] or title_from_list
        publish_date = detail["date"]
        content = detail["content"]
        source = detail.get("source", "兖州区生态环境局")
        
        if not content:
            # Attach embed for short content
            content = f"[无正文内容] {title}"
        
        batch.append({
            "url": url,
            "title": title,
            "content": content,
            "pub_date": publish_date,
            "site_name": "兖州区人民政府-建设项目环境影响评价",
            "source_url": url,
        })
        
        print(f"  [{idx+1}] {title[:45]}... {publish_date}")
        time.sleep(0.3)
    
    if batch:
        result = push_to_searchdb(batch)
        print(f"\n[DONE] 完成: {len(batch)} 条")
    else:
        print("[DONE] 无数据")

if __name__ == "__main__":
    main()
