#!/usr/bin/env python3
"""Explore sinodmc.com list structure"""
import sys, re
from bs4 import BeautifulSoup
import urllib.request, ssl

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

req = urllib.request.Request("https://www.sinodmc.com/list.php?catid=65&page=1",
    headers={"User-Agent": "Mozilla/5.0"})
resp = urllib.request.urlopen(req, timeout=30, context=ssl_ctx)
html = resp.read().decode("utf-8")

soup = BeautifulSoup(html, "html.parser")

rows = soup.find_all("div", class_="row")
for row in rows:
    cols = row.find_all("div", class_="col-md-6")
    if not cols:
        continue
    for col in cols:
        newslist = col.find("div", class_="newslist")
        if not newslist:
            continue
        for child in newslist.find_all(recursive=False):
            a = child.find("a", href=re.compile(r"show\.php"))
            if not a:
                continue
            title = a.get_text(strip=True)
            if not title:
                continue
            href = a["href"]
            if not href.startswith("http"):
                href = "https://www.sinodmc.com" + href if href.startswith("/") else "https://www.sinodmc.com/" + href
            
            spans = child.find_all("span")
            date = spans[0].get_text(strip=True) if spans else ""
            
            print(f"{href[:60]} | {title[:60]} | {date}")
        
        # Check for any content in the newslist
        texts = newslist.get_text(" ", strip=True)
        dates_in_list = re.findall(r"\d{4}\.\d{2}\.\d{2}", texts)
        if dates_in_list:
            print(f"  Sample dates: {dates_in_list[:5]}")
        break
    break

# Check detail page
print("\n=== Detail page ===")
req2 = urllib.request.Request("https://www.sinodmc.com/show.php?contentid=2314",
    headers={"User-Agent": "Mozilla/5.0"})
resp2 = urllib.request.urlopen(req2, timeout=30, context=ssl_ctx)
html2 = resp2.read().decode("utf-8", errors="replace")
soup2 = BeautifulSoup(html2, "html.parser")

# Title
for h in soup2.find_all(["h1","h2","h3","h4","title"]):
    txt = h.get_text(strip=True)
    if len(txt) > 10:
        print(f"{h.name}.{h.get('class','')}: {txt[:80]}")

# Date pattern
m = re.search(r"(\d{4}[-/.]\d{1,2}[-/.]\d{1,2})", html2)
if m:
    print(f"Date found: {m.group(1)}")

# Content area
for cls_name in ["content", "main", "article", "text", "body", "detail"]:
    div = soup2.find("div", class_=lambda x: x and cls_name in x.lower()) if cls_name else None
    if div:
        txt = div.get_text(" ", strip=True)[:100]
        print(f"Div with '{cls_name}': {txt}")
        break

# Check text length
text_blocks = soup2.find_all(["p","div"], class_=lambda x: x and "content" in x.lower())
for tb in text_blocks[:3]:
    txt = tb.get_text(" ", strip=True)
    print(f"Content block: {txt[:100]}... (len={len(txt)})")
