import sys, re

html = sys.stdin.read()

print(f"HTML size: {len(html)}")
print()

# All public/239 links
links = re.findall(r'href=["\'](/public/239/[^"\']+)["\']', html)
print(f"Links to /public/239/: {len(links)}")
for l in links[:10]:
    print(f"  {l}")

# All public/ links
all_links = re.findall(r'href=["\'](/public/\d+/\d+\.html)["\']', html)
print(f"\nAll article links: {len(all_links)}")
for l in all_links[:10]:
    print(f"  {l}")

# Check what data is in the page
if 'var ll_' in html:
    print("\nContains JavaScript variable declarations")
if 'ajax' in html.lower():
    print("Contains AJAX calls")

# Check if template variables present
if '<?=' in html or '{{' in html:
    print("Page has unrendered template variables")

# Check for a content container
# Look at main content area
for m in re.finditer(r'<div[^>]*class=["\']([^"\']*(?:content|list|article|main)[^"\']*)["\'][^>]*>(.*?)</div>', html, re.S):
    txt = re.sub(r'<[^>]+>', ' ', m.group(2))
    txt = re.sub(r'\s+', ' ', txt).strip()
    if len(txt) > 50:
        print(f"\nContent div .{m.group(1)}: {txt[:200]}")

# Look for the actual list items 
items = re.findall(r'<li[^>]*>.*?<a[^>]*href=["\']([^"\']+)["\'][^>]*>(.*?)</a>.*?</li>', html, re.S)
print(f"\nLi with links: {len(items)}")
for href, a_html in items[:5]:
    txt = re.sub(r'<[^>]+>', ' ', a_html).strip()
    txt = re.sub(r'\s+', ' ', txt).strip()
    print(f"  {txt[:100]} -> {href}")
