#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""石大胜华新材料集团 - 信息公开爬虫"""
import sys, re, os, json, argparse
sys.path.insert(0, "/root")
sys.path.insert(0, "/root/gov_crawler")
from urllib.parse import urljoin
from datetime import datetime
import urllib.request, ssl

SITE_NAME = "石大胜华新材料集团-信息公开"
BASE_URL = "https://www.sinodmc.com"
LIST_URL = "https://www.sinodmc.com/list.php?catid=65"

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)

def fetch_page(page_url):
    req = urllib.request.Request(page_url, headers={"User-Agent": "Mozilla/5.0"})
    resp = urllib.request.urlopen(req, timeout=30, context=ssl_ctx)
    return resp.read().decode("utf-8")


def parse_list_page(html):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    items = []

    for row in soup.find_all("div", class_="row"):
        for col in row.find_all("div", class_="col-md-6"):
            newslist = col.find("div", class_="newslist")
            if not newslist:
                continue
            for child in newslist.find_all(recursive=False):
                a = child.find("a", href=re.compile(r"show\.php"))
                if not a:
                    continue
                title = a.get_text(strip=True)
                if not title or len(title) < 5:
                    continue
                href = a["href"]
                if not href.startswith("http"):
                    href = urljoin(BASE_URL, href)

                # Find date from span        
                spans = child.find_all("span")
                date_str = ""
                for sp in spans:
                    t = sp.get_text(strip=True)
                    m = re.search(r"(\d{4})[./-](\d{1,2})[./-](\d{1,2})", t)
                    if m:
                        date_str = f"{m.group(1)}-{m.group(2).zfill(2)}-{m.group(3).zfill(2)}"
                        break

                items.append({
                    "url": href,
                    "title": title,
                    "date": date_str,
                })
    return items


def get_next_page_url(html, current_page):
    """从分页器中找下一页"""
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    # Look for page=2, page=3 etc links
    next_page = current_page + 1
    for a in soup.find_all("a", href=re.compile(rf"catid=65&page={next_page}")):
        return urljoin(LIST_URL, a["href"])
    return None


def fetch_detail(url):
    try:
        req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
        resp = urllib.request.urlopen(req, timeout=30, context=ssl_ctx)
        html = resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        return "", "", [], ""

    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")

    # Title from <title>
    title = ""
    m = re.search(r"<title>(.*?)</title>", html)
    if m:
        t = m.group(1)
        t = t.replace("_信息公开_石大胜华", "").replace("_石大胜华", "").strip()
        if t:
            title = t

    # Title from h3
    h3 = soup.find("h3")
    if h3 and not title:
        title = h3.get_text(strip=True)

    # Date
    date_str = ""
    for pat in [
        r"时间[：:]\s*(\d{4}[-/.]\d{1,2}[-/.]\d{1,2})",
        r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})",
        r"(\d{4}-\d{2}-\d{2})",
    ]:
        m = re.search(pat, html)
        if m:
            date_str = m.group(1).replace("/", "-").replace(".", "-")[:10]
            break

    # Content
    content_div = soup.find("div", class_=lambda x: x and "content" in str(x).lower() and "nav" not in str(x).lower())
    if not content_div or len(content_div.get_text(strip=True)) < 50:
        # Try the article body
        content_div = soup.find("div", class_=lambda x: x and ("detail" in str(x).lower() or "article" in str(x).lower() or "text" in str(x).lower()))
    if not content_div or len(content_div.get_text(strip=True)) < 50:
        content_div = soup.find("div", class_="container")
        if content_div:
            # Skip first few children (nav/menu)
            for child in content_div.find_all(recursive=False):
                if child.name == "div" and len(child.get_text(strip=True)) > 200:
                    content_div = child
                    break

    content = ""
    attachments = []

    if content_div:
        for tag in content_div.find_all(["script", "style"]):
            tag.decompose()

        parts = []
        for el in content_div.find_all(["p", "table", "img"]):
            if el.name == "p" and not el.find_parent("table") and not el.find("table") and not el.find("p"):
                txt = el.get_text(" ", strip=True)
                txt = re.sub(r"\s+", " ", txt)
                txt = re.sub(r"(?<=[\u4e00-\u9fff])[ \t\u3000]+(?=[\u4e00-\u9fff])", "", txt)
                if txt and len(txt) > 5:
                    parts.append(txt)
            if el.name == 'table':
                parts.append(html_table_to_html(el, url))
            elif el.name == "img":
                src = el.get("src", "")
                alt = el.get("alt", "图片")
                if src and not src.startswith("data:"):
                    parts.append(f'<p><a href="{urljoin(url, src)}">查看图片</a></p>')

        content = "\n\n".join(parts)

        for a_tag in content_div.find_all("a", href=True):
            href = a_tag["href"]
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", href, re.I):
                full_url = urljoin(url, href)
                name = a_tag.get_text(strip=True) or os.path.basename(href)
                attachments.append({"name": name, "url": full_url})

    if attachments:
        if content:
            content += "\n\n"
        att_lines = [f"[{att['name']}]({att['url']})" for att in attachments]
        content += "\n".join(att_lines)

    return title, date_str, attachments, content


def main():
    parser = argparse.ArgumentParser(description="石大胜华信息公开爬虫")
    parser.add_argument("--pages", type=int, default=5)
    args = parser.parse_args()

    pages = args.pages
    print(f"=== {SITE_NAME} ===")
    print(f"  pages: {pages}")

    all_items = []
    for page in range(1, pages + 1):
        url = f"{LIST_URL}&page={page}"
        print(f"  [Page {page}/{pages}] fetching...", end=" ")
        try:
            html = fetch_page(url)
            items = parse_list_page(html)
            print(f"{len(items)} items")
            all_items.extend(items)
        except Exception as e:
            print(f"ERROR: {e}")
            break

    print(f"\n=== Total {len(all_items)} items ===")

    seen = set()
    unique = []
    for item in all_items:
        if item["url"] not in seen:
            seen.add(item["url"])
            unique.append(item)
    print(f"Unique: {len(unique)}")

    db_items = []
    for i, item in enumerate(unique):
        print(f"  [{i+1}/{len(unique)}] {item['title'][:40]}...", end=" ")
        title, date_str, attachments, content = fetch_detail(item["url"])
        use_title = title or item["title"]
        use_date = date_str or item.get("date", "")
        summary = re.sub(r"\s+", "", content)[:200] if content else ""

        db_items.append({
            "site_name": SITE_NAME,
            "source_url": item["url"],
            "url": item["url"],
            "title": use_title,
            "pub_date": use_date,
            "summary": summary,
            "content": content,
            "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        })
        print(f"ok [{use_date}]")

    try:
        from crawler_lib import push_to_searchdb
        push_to_searchdb(db_items, SITE_NAME)
        print(f"\nOK: {len(db_items)}")
    except Exception as e:
        print(f"FAIL: {e}")
        import traceback; traceback.print_exc()


if __name__ == "__main__":
    main()
