#!/usr/bin/env python3
"""Search nanpu API across all pages for 唐山晟红"""
import requests, json, sys
sys.stdout.reconfigure(encoding='utf-8')

headers = {
    "User-Agent": "Mozilla/5.0",
    "Referer": "https://www.nanpu.gov.cn/news/18/",
    "Accept": "application/json, text/plain, */*"
}
api_url = "https://www.nanpu.gov.cn/fwebapi/cms/lowcode/60003/18505/list?cate=0"
PAGE_SIZE = 200

for page in range(10):
    payload = {
        "size": PAGE_SIZE,
        "query": [{
            "esField": "DETAIL_ES.es_multi_category_6d5k7017",
            "groupEnd": "1",
            "field": "category_6d5k7017",
            "sourceType": "page",
            "dataType": "array[category]",
            "logic": "and",
            "groupBegin": "1",
            "value": "1628181411606835200",
            "operator": "in"
        }],
        "from": page * PAGE_SIZE,
        "sort": [],
        "_detailId": "1628181411606835200"
    }

    r = requests.post(api_url, json=payload, headers=headers, timeout=30)
    data = r.json()

    if data.get("status") == "200":
        items = data["data"].get("list", [])
        if not items:
            print(f"Page {page+1}: empty, stopping")
            break
        print(f"Page {page+1}: {len(items)} items [{items[0].get('title','')[:40]}...{items[-1].get('title','')[:40]}]")
        for item in items:
            title = item.get("title", "")
            if "晟红" in title or "2,4-滴" in title or "滴·氨氯" in title:
                print(f"\nFOUND on page {page+1}!")
                print(f"Title: {title}")
                item_id = item.get("id", "")
                detail_url = f"https://www.nanpu.gov.cn/zwxx/{item_id}.html"
                print(f"Detail URL: {detail_url}")
                
                # Fetch detail to check content type
                r2 = requests.get(detail_url, headers=headers, timeout=30)
                r2.encoding = 'utf-8'
                from bs4 import BeautifulSoup
                soup = BeautifulSoup(r2.text, 'html.parser')
                rt = soup.find(class_='e_richText-11')
                if rt:
                    imgs = rt.find_all('img')
                    ps = rt.find_all('p')
                    txt = rt.get_text(strip=True)
                    print(f"\nContent in .e_richText-11:")
                    print(f"  Text: {len(txt)} chars")
                    print(f"  Images: {len(imgs)}")
                    print(f"  P-tags with text: {len([p for p in ps if p.get_text(strip=True)])}")
                    for p in ps:
                        ptxt = p.get_text(strip=True)
                        if ptxt and len(ptxt) > 5:
                            print(f"  <p>: {ptxt[:200]}")
                    for img in imgs[:3]:
                        print(f"  <img>: {img.get('src','')[:60]} ({img.get('alt','')[:30]})")
                else:
                    print("No .e_richText-11 found!")
                sys.exit(0)
    else:
        print(f"Page {page+1}: API error")
        break
else:
    print("\nNot found in first 2000 items")
