#!/usr/bin/env python3
"""
ynfushi.com (云南云天化福石科技) — 通知公告 爬虫
CMS: 自定义PHP (Powered by aykj.net)
WAF: 云锁 (Yunsuo) — 需4步绕过
列表: /list/cnPc/22/45/auto/12/0.html (第1页)
AJAX: /subsiteIndex/toPage?subsiteFlag=cnPc&subsiteId=22&newsClassId=45&pageType=auto&pageSize=12&start=N
详情: /view/cnPc/22/45/view/{id}.html
"""

import requests, re, sys, os, time
from bs4 import BeautifulSoup

BASE = "http://www.ynfushi.com"
LIST_URL = BASE + "/list/cnPc/22/45/auto/12/0.html"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
}

CRAWLER_DIR = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, CRAWLER_DIR)
from crawler_lib import push_to_searchdb

MAX_PAGES = int(sys.argv[sys.argv.index("--pages") + 1]) if "--pages" in sys.argv else 5


def str2hex(s):
    return "".join(hex(ord(c))[2:] for c in s)


def create_session():
    """Create an authenticated session (Yunsuo 4-step bypass)"""
    sess = requests.Session()
    sess.headers.update(HEADERS)
    url = LIST_URL
    try:
        sess.get(url, timeout=30)
    except:
        pass
    sess.cookies.set("srcurl", str2hex(url), domain="www.ynfushi.com", path="/")
    hex_data = str2hex("1920,1080")
    try:
        sess.get(url + "?security_verify_data=" + hex_data, timeout=30)
    except:
        pass
    try:
        sess.get(url, timeout=30)
    except:
        pass
    return sess


def fetch_ajax_page(sess, page_num):
    """Fetch one page of the list via AJAX endpoint"""
    url = (BASE +
        "/subsiteIndex/toPage?subsiteFlag=cnPc&subsiteId=22"
        "&newsClassId=45&pageType=auto&pageSize=12"
        "&start=%d&objectId=") % page_num
    try:
        r = sess.get(url, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print("[ERROR] fetch page %d: %s" % (page_num, e))
        return ""


def parse_list(html):
    """Parse list items from AJAX response"""
    soup = BeautifulSoup(html, "html.parser")
    items = []
    for con in soup.find_all("div", class_="con"):
        title_div = con.find("div", class_="title")
        time_div = con.find("div", class_="time")
        if not title_div or not time_div:
            continue
        a = title_div.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href.startswith("http"):
            href = BASE + href
        title = a.get("title", "") or a.get_text(strip=True)
        pub_date = time_div.get_text(strip=True)
        items.append({"title": title, "url": href, "pub_date": pub_date})
    return items


def fetch_detail(sess, url):
    """Fetch article detail and extract content"""
    try:
        r = sess.get(url, timeout=30)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print("[ERROR] fetch_detail %s: %s" % (url, e))
        return None

    soup = BeautifulSoup(html, "html.parser")

    # Extract real title from <title> tag (full text, never truncated)
    # Format: "真实标题|云南云天化福石科技有限公司官网|..."
    full_title = ""
    title_tag = soup.find("title")
    if title_tag:
        raw = title_tag.get_text(strip=True)
        parts = raw.split("|")
        if parts and parts[0].strip():
            full_title = parts[0].strip()

    # Content extraction
    content_div = soup.find("div", class_="articleBox")
    if not content_div:
        content_div = soup.find("div", class_="articleC")
    if not content_div:
        print("[WARN] No content div found: %s" % url)
        return None

    parts = []
    for elem in content_div.find_all(["p", "table"], recursive=True):
        if elem.name == "p":
            if elem.find_parent("table"):
                continue
            text = elem.get_text(strip=True)
            if text:
                parts.append(text)
        elif elem.name == "table":
            parts.append(str(elem))

    content = "\n\n".join(parts) if parts else ""
    if not content:
        content = "[无正文内容]"

    # Fallback date from detail page
    date = ""
    m = re.search(r'发布时间[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if m:
        date = m.group(1)

    return {"content": content, "date_from_detail": date, "full_title": full_title}


def main():
    sess = create_session()

    # Check WAF bypass success
    html_p1 = fetch_ajax_page(sess, 1)
    if not html_p1 or len(html_p1) < 2000:
        print("[ERROR] Cannot fetch page 1 - WAF bypass may have failed")
        sys.exit(1)

    items_p1 = parse_list(html_p1)
    print("[INFO] Page 1: %d items" % len(items_p1))

    # Try pages 2-3 to find total
    total_pages = 1
    for pg in [2, 3]:
        html = fetch_ajax_page(sess, pg)
        if html and len(html) > 2000:
            parsed = parse_list(html)
            if parsed:
                total_pages = pg
        else:
            break
    print("[INFO] Total pages: %d" % total_pages)

    all_items = list(items_p1)
    pages_to_fetch = min(total_pages, MAX_PAGES)
    for page in range(2, pages_to_fetch + 1):
        html = fetch_ajax_page(sess, page)
        if html and len(html) > 2000:
            items = parse_list(html)
            if items:
                all_items.extend(items)
                print("[INFO] Page %d: %d items" % (page, len(items)))
        time.sleep(0.5)

    print("[INFO] Total items: %d" % len(all_items))

    batch = []
    for idx, item in enumerate(all_items):
        title = item["title"]
        url = item["url"]
        pub_date = item["pub_date"]

        detail = fetch_detail(sess, url)
        if detail is None:
            continue

        content = detail["content"]
        # Use <title> tag as full title (never truncated)
        if detail.get("full_title"):
            title = detail["full_title"]
        # Use detail date as fallback
        if not pub_date and detail.get("date_from_detail"):
            pub_date = detail["date_from_detail"]

        batch.append({
            "url": url,
            "title": title,
            "content": content,
            "pub_date": pub_date,
            "site_name": "云南云天化福石-通知公告",
            "source_url": url,
        })

        print("  [%d] %s... %s" % (idx + 1, title[:45], pub_date))
        time.sleep(0.5)

    if batch:
        result = push_to_searchdb(batch)
        print("\n[DONE] 完成: %d 条" % len(batch))
    else:
        print("[DONE] 无数据")


if __name__ == "__main__":
    main()
