#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
云南祥丰实业集团 - 最新公告 (环评公示) 爬虫
CMS: 自定义PHP
WAF: 云锁 YunSuo (JS challenge + cookie)
列表: /list/xfhfPC/1/16/auto/20/{page}.html (0-based, 20条/页)
详情: /view/xfhfPC/1/16/view/{id}.html
详情标题: <title>
详情正文: div.articleBox > p
详情日期: 2026年07月09日 (from div.subCon or regex)

用法:
  python3 crawl_xfhf.py
  python3 crawl_xfhf.py --pages 3
"""

import re, sys, time, requests, sqlite3
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.xfhf.com"
LIST_URL = BASE_URL + "/list/xfhfPC/1/16/auto/20/{}.html"
DB_PATH = "/root/search.db"
SITE_NAME = "云南祥丰-最新公告"
CATEGORY = GROUP = "环评"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
MAX_PAGES = 1
TIMEOUT = 30


def str2hex(s):
    return "".join(hex(ord(c))[2:] for c in s)


def bypass_request(url):
    """Bypass YunSuo WAF - 4 step flow"""
    sess = requests.Session()
    sess.headers.update(HEADERS)
    sess.get(url, timeout=TIMEOUT)
    sess.cookies.set("srcurl", str2hex(url), domain="www.xfhf.com", path="/")
    hex_data = str2hex("1366,768")
    sess.get(url + "?security_verify_data=" + hex_data, timeout=TIMEOUT)
    r = sess.get(url, timeout=TIMEOUT)
    return r.text


def fetch(url):
    for attempt in range(3):
        try:
            return bypass_request(url)
        except Exception as e:
            if attempt == 2:
                print(f"  [WARN] 获取失败 ({attempt+1}/3): {url} - {e}", file=sys.stderr)
                return ""
            time.sleep(2)


def extract_list_items(html):
    items = []
    soup = BeautifulSoup(html, "html.parser")
    subpage = soup.find("div", class_="subPageRC")
    if not subpage:
        return items
    for wrap in subpage.find_all("div", class_="wrap"):
        a_tag = wrap.select_one("div.font a")
        if not a_tag:
            continue
        href = a_tag.get("href", "")
        title = a_tag.get_text(strip=True)
        if not href or not title:
            continue
        full_url = urljoin(BASE_URL, href)
        time_tag = wrap.select_one("div.time")
        date = time_tag.get_text(strip=True) if time_tag else ""
        items.append((full_url, title, date))
    return items


def extract_detail(html):
    soup = BeautifulSoup(html, "html.parser")

    title = ""
    title_tag = soup.find("title")
    if title_tag and title_tag.string:
        title = title_tag.string.strip()
        title = re.sub(r"\|云南祥丰实业集团有限公司$", "", title).strip()

    date = ""
    subcon = soup.find("div", class_="subCon")
    if subcon:
        m = re.search(r"(\d{4})年(\d{2})月(\d{2})日", subcon.get_text())
        if m:
            date = "%s-%s-%s" % (m.group(1), m.group(2), m.group(3))
    if not date:
        m = re.search(r"(\d{4})年(\d{2})月(\d{2})日", html)
        if m:
            date = "%s-%s-%s" % (m.group(1), m.group(2), m.group(3))

    content = ""
    article = soup.find("div", class_="articleBox")
    if article:
        parts = []
        for p in article.find_all("p"):
            text = p.get_text(strip=True)
            if text:
                parts.append(text)
        content = "\n\n".join(parts)

    attachments = []
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if ".pdf" in href.lower():
            name = a.get_text(strip=True) or href.split("/")[-1]
            attachments.append({"name": name, "url": urljoin(BASE_URL, href)})

    return title, date, content, attachments


def push_to_db(conn, items):
    saved = 0
    skipped = 0
    for url, title, date, content, attachments in items:
        import json
        attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
        summary = content[:200].replace("\n", " ") if content else title
        if not content and attachments:
            content = "\n".join(f"[{a['name']}]({a['url']})" for a in attachments)
        try:
            cur = conn.execute(
                "INSERT OR REPLACE INTO gov_raw "
                "(page_url, title, site_name, publish_date, content, date_rank, summary, attachments) "
                "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                (url, title, SITE_NAME, date, content, date, summary, attachments_json)
            )
            if cur.rowcount > 0:
                saved += 1
            else:
                skipped += 1
        except Exception as e:
            print(f"  [ERR] DB: {url} - {e}", file=sys.stderr)
            skipped += 1
    return saved, skipped


def main():
    pages = MAX_PAGES
    for i, a in enumerate(sys.argv):
        if a == "--pages" and i + 1 < len(sys.argv):
            pages = int(sys.argv[i + 1])
            break

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    all_items = []
    print(f"云南祥丰-最新公告: 爬取 {pages} 页")
    for idx in range(pages):
        list_url = LIST_URL.format(idx)
        print(f"  [列表页 {idx+1}/{pages}] {list_url}")
        html = fetch(list_url)
        if not html:
            continue
        items = extract_list_items(html)
        if not items:
            print("    -> 无数据，停止")
            break
        print(f"    -> 找到 {len(items)} 条")
        for item_url, item_title, item_date in items:
            time.sleep(0.5)
            print(f"    [{len(all_items)+1}] {item_title[:40]}...", end=" ", flush=True)
            detail_html = fetch(item_url)
            if not detail_html:
                print("fail")
                continue
            title, date, content, attachments = extract_detail(detail_html)
            if not title:
                title = item_title
            if not date:
                date = item_date
            all_items.append((item_url, title, date, content, attachments))
            status = "ok" if content else "empty"
            attach_msg = f" +{len(attachments)}pdf" if attachments else ""
            print(f"({status}, {len(content)}字{attach_msg})")
    if all_items:
        saved, skipped = push_to_db(conn, all_items)
        conn.commit()
        print(f"\n完成! 新增: {saved}, 跳过: {skipped}, 总共: {len(all_items)}")
    conn.close()
    return 0


if __name__ == "__main__":
    sys.exit(main())
