#!/usr/bin/env python3
"""
Crawler for 仪陇县人民政府 - 环评公示
https://www.yilong.gov.cn/zwgk/fdzdgknr/hjbh_7351/hpgs/
CMS: TRS (TRS_UEDITOR)
"""
import sys, os, re, time
from datetime import datetime, date
from urllib.parse import urljoin

sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import fetch_page, push_to_searchdb

SITE_NAME = "yilong.gov.cn-环评公示"
BASE_URL = "https://www.yilong.gov.cn/zwgk/fdzdgknr/hjbh_7351/hpgs/"
CATEGORY = "环评公示"
CUTOFF = date(2023, 6, 16)


def extract_content(html):
    """Extract article body from xl-article div using BeautifulSoup."""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, 'html.parser')
    
    article = soup.find("div", class_="xl-article")
    if not article:
        return ""
    
    # Remove scripts and styles
    for tag in article.find_all(['script', 'style']):
        tag.decompose()
    
    # TRS editor view is the main content
    trs = article.find("div", class_=re.compile(r"trs_editor_view|TRS_UEDITOR"))
    if trs:
        return str(trs)
    
    # Fallback: all non-empty content
    return str(article)


def get_meta(html, name):
    m = re.search(r'<meta[^>]*name="' + name + r'"[^>]*content="(.*?)"', html)
    return m.group(1).strip() if m else ""


def parse_ymd(s):
    """Try parsing YYYY-MM-DD or YYYY/MM/DD."""
    for fmt in ["%Y-%m-%d", "%Y/%m/%d"]:
        try:
            return datetime.strptime(s.strip(), fmt).date()
        except:
            pass
    return None


def extract_page_items(page_url):
    """Return list of (full_url, title, list_date_str)."""
    items = []
    html = fetch_page(page_url)
    if not html:
        return items
    
    m = re.search(r'<ul>\s*(.*?)\s*</ul>', html[html.find("xxgk-list"):], re.DOTALL)
    if not m:
        return items
    
    ul_html = m.group(1)
    for li in re.finditer(r'<li[^>]*>(.*?)</li>', ul_html, re.DOTALL):
        content = li.group(1)
        a = re.search(r'<a[^>]*href="(.*?)"[^>]*>(.*?)</a>', content, re.DOTALL)
        if not a:
            continue
        href = a.group(1).strip()
        title = re.sub(r'<[^>]+>', '', a.group(2)).strip()
        if not title:
            continue
        span = re.search(r'<span>(.*?)</span>', content, re.DOTALL)
        list_date = span.group(1).strip() if span else ""
        items.append((urljoin(page_url, href), title, list_date))
    
    return items


def main():
    print(f"站点: {SITE_NAME}  截止: {CUTOFF}\n")
    
    # Collect
    print("[列表页...]")
    all_items = []
    for p in [BASE_URL, urljoin(BASE_URL, "index_1.html"), urljoin(BASE_URL, "index_2.html")]:
        its = extract_page_items(p)
        print(f"  {p}: {len(its)} 条")
        all_items.extend(its)
        if not its:
            break
    print(f"\n总计: {len(all_items)} 条")
    
    # Process
    print("\n[详情页...]")
    batch = []
    skip_count = 0
    
    for i, (href, title, list_date) in enumerate(all_items, 1):
        print(f"\n[{i}/{len(all_items)}]", end=" ")
        
        dt = parse_ymd(list_date)
        is_doc = bool(re.search(r'\.(doc|docx|pdf)$', href, re.I))
        
        if is_doc:
            if dt and dt >= CUTOFF:
                entry = {
                    "site_name": SITE_NAME,
                    "title": title,
                    "url": href,
                    "source_url": href,
                    "pub_date": str(dt),
                    "content": f'<p><a href="{href}">{os.path.basename(href)}</a></p>',
                    "summary": title[:300],
                    "category": CATEGORY,
                    "tags": "",
                }
                batch.append(entry)
                print(f"[OK] doc: {title[:40]}...")
            else:
                print(f"[SKIP] doc dt={dt}")
                skip_count += 1
            continue
        
        # HTML page
        html = fetch_page(href)
        if not html:
            print("[ERROR] fetch fail")
            skip_count += 1
            continue
        
        # Meta from detail page
        pub_date_str = get_meta(html, "PubDate")[:10]
        dt = parse_ymd(pub_date_str) if pub_date_str else dt  # detail date > list date
        
        if dt is None:
            print(f"[SKIP] no date")
            skip_count += 1
            continue
        if dt < CUTOFF:
            print(f"[SKIP] {dt}")
            skip_count += 1
            continue
        
        detail_title = get_meta(html, "ArticleTitle") or title
        source = get_meta(html, "ContentSource")
        content = extract_content(html)
        
        # Attachments
        att_m = re.search(r"var\s+hasFJ\s*=\s*'(.*?)'", html, re.DOTALL)
        if att_m and att_m.group(1):
            attach_html = att_m.group(1)
            # Convert relative URLs
            attach_html = re.sub(r'href="(?!https?://)', f'href="{href.rsplit("/", 1)[0]}/', attach_html)
            content += f'\n<div class="attachments">{attach_html}</div>'
        
        summary = re.sub(r'<[^>]+>', ' ', content)
        summary = re.sub(r'\s+', ' ', summary).strip()[:300]
        
        entry = {
            "site_name": SITE_NAME,
            "title": detail_title,
            "url": href,
            "source_url": href,
            "pub_date": str(dt),
            "content": content,
            "summary": summary,
            "category": CATEGORY,
            "tags": "",
        }
        batch.append(entry)
        print(f"[OK] {detail_title[:40]}...")
        time.sleep(0.3)
    
    print(f"\n{'='*50}")
    print(f"入库: {len(batch)}, 跳过: {skip_count}")
    if batch:
        push_to_searchdb(batch)
    print(f"{'='*50}")


if __name__ == "__main__":
    main()
