import sys, re, html as html_mod

body = sys.stdin.read()

# Title tag
t = re.search(r'<title>(.*?)</title>', body, re.S)
print('TITLE TAG:', t.group(1) if t else 'N/A')

# h1
for m in re.findall(r'<h1[^>]*>(.*?)</h1>', body, re.S):
    txt = re.sub(r'<[^>]+>', ' ', m)
    txt = re.sub(r'\s+', ' ', txt).strip()
    print('H1:', txt[:150])

# Content classes
for m in re.findall(r'class=["\']([^"\']*(?:content|article|text|TRS|nr|main|detail|xxgk)[^"\']*)["\']', body):
    print('CONTENT CLASS:', m)

# Date
for m in re.findall(r'\d{4}-\d{2}-\d{2}', body):
    print('DATE:', m)
    break

# Look for the main content block (text-rich div)
for m in re.finditer(r'<(div|section)[^>]*>(.{200,}?)</\1>', body, re.S):
    txt = re.sub(r'<[^>]+>', ' ', m.group(2))
    txt = re.sub(r'\s+', ' ', txt).strip()
    if len(txt) > 100:
        cls = re.search(r'class=["\']([^"\']*)["\']', m.group(0))
        print(f'BLOCK <{m.group(1)}> class={cls.group(1) if cls else "none"}: len={len(txt)}')
        print(f'  TEXT: {txt[:200]}')
