import json
import re
from collections import defaultdict

# Load config
with open('/root/gov_crawler/daily_crawl_config.json', 'r') as f:
    configs = json.load(f)

# Extract remaining names from target groups
target_groups = ['环评', '政府公告', '区县', '其他', '生态环境']
remaining = []
for c in configs:
    g = c.get('group','')
    if g in target_groups:
        n = c.get('name','')
        if n not in remaining:
            remaining.append(n)

print(f'Total remaining entries: {len(remaining)}')
print()

# Now let's see what place names are in these names
# Extract potential place names (2-4 Chinese chars at start)
potential = []
for n in remaining:
    # Extract the first meaningful word before any dash/hyphen
    m = re.match(r'^([\u4e00-\u9fff]{2,6})', n)
    if m:
        potential.append(m.group(1))

# Count frequency
from collections import Counter
freq = Counter(potential)
for name, count in freq.most_common():
    print(f'  {name}({count})')
