#!/usr/bin/env python3
"""Find the API endpoint and data structure for guzhen.gov.cn"""
import re
from bs4 import BeautifulSoup

with open('/root/gov_crawler/guzhen_page.html', 'r', encoding='utf-8') as f:
    html = f.read()

soup = BeautifulSoup(html, 'html.parser')

# 1. Find the actual rendered list content
# Look for table.xxgk-table-list
tables = soup.find_all('table', class_=lambda x: x and 'xxgk-table-list' in str(x) if x else False)
print('Found %d xxgk-table-list tables' % len(tables))
for t in tables[:2]:
    rows = t.find_all('tr')
    print('  Rows: %d' % len(rows))
    for r in rows[:5]:
        print('  %s' % r.text.strip()[:100].replace('\n', ' '))

# 2. Look for API URLs in scripts
for script in soup.find_all('script'):
    if script.string:
        # Find API calls
        urls = re.findall(r'[\"\'](/[^\"\']*(?:label|api|list|data|ajax|getData|loadData)[^\"\']*)[\"\']', script.string, re.I)
        if urls:
            print('\nAPI URLs found in script:')
            for u in urls[:10]:
                print('  %s' % u)

# 3. Check for any JSON data embedded
for script in soup.find_all('script'):
    if script.string and ('data:' in script.string or 'list:' in script.string or 'items:' in script.string):
        print('\nScript with data: %s...' % script.string[:200].replace('\n', ' '))

# 4. Check for the ls_label or site_label pattern
for m in re.finditer(r'/zfxxgk/site/label/\d+', html):
    print('\nLabel URL: %s' % m.group())

# 5. Search for hidden input with total count or page info
for inp in soup.find_all('input', type='hidden'):
    name = inp.get('name', '')
    val = inp.get('value', '')
    if name and val and any(x in name.lower() for x in ['page', 'total', 'count', 'size']):
        print('\nHidden input: %s = %s' % (name, val))

# 6. Look for the actual list generation script
# Epoint WebBuilder typically has a template like:
# <script id="list_tpl" type="text/html">...</script>
for tag in soup.find_all('script', type=lambda x: x and 'text/html' in x if x else False):
    content = tag.string or ''
    if 'data' in content.lower() and ('el.' in content or 'el[' in content):
        print('\nTemplate script (id=%s):' % tag.get('id', ''))
        print(content[:500])

# 7. Look for the config object with page/pageSize
print('\n--- Looking for API init code ---')
for s in soup.find_all('script'):
    if s.string and ('siteId' in s.string and 'pageSize' in s.string):
        print(s.string[:500])
        break
