#!/usr/bin/env python3
"""Check if 唐山晟红 page has text content"""
import sqlite3
import requests
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
conn = sqlite3.connect(DB_PATH)
c = conn.cursor()

c.execute("SELECT page_url, length(content), substr(content,1,200) FROM gov_raw WHERE title LIKE '%唐山晟红%'")
row = c.fetchone()
if row:
    print(f"URL: {row[0]}")
    print(f"Content length: {row[1]}")
    print(f"Content: {row[2]}")
    url = row[0]
else:
    print("未在DB中找到（可能不在已爬数据中）。从列表搜索相近标题...")
    c.execute("SELECT page_url, title FROM gov_raw WHERE page_url LIKE '%nanpu%' AND title LIKE '%环%' ORDER BY id DESC LIMIT 10")
    found = c.fetchall()
    for r in found:
        print(f"  {r[1][:60]} → {r[0]}")
    url = None

conn.close()

if not url:
    # Search on live site
    print("\nLive site搜索...")
    headers = {"User-Agent": "Mozilla/5.0"}
    s = requests.Session()
    s.headers.update(headers)
    
    # First get list page
    r = s.get("https://www.nanpu.gov.cn/news/18/", timeout=30)
    r.encoding = "utf-8"
    soup = BeautifulSoup(r.text, "html.parser")
    
    # Find all links
    for a in soup.find_all("a", href=True):
        txt = a.get_text(strip=True)
        if "晟红" in txt or "2,4-滴" in txt:
            href = a["href"]
            full_url = href if href.startswith("http") else f"https://www.nanpu.gov.cn{href}"
            print(f"\n找到: {txt[:80]}")
            print(f"URL: {full_url}")
            
            # Fetch detail
            r2 = s.get(full_url, timeout=30)
            r2.encoding = "utf-8"
            soup2 = BeautifulSoup(r2.text, "html.parser")
            
            rt = soup2.find(class_="e_richText-11")
            if rt:
                imgs = rt.find_all("img")
                figs = rt.find_all("figure")
                ps = rt.find_all("p")
                txt_content = rt.get_text(strip=True)
                print(f"\n.e_richText-11: {len(txt_content)} chars, {len(imgs)} imgs, {len(figs)} figures, {len(ps)} p-tags")
                for p in ps:
                    ptxt = p.get_text(strip=True)
                    if ptxt:
                        print(f"  <p>: {ptxt[:100]}")
                for img in imgs[:3]:
                    src = img.get("src", "")[:80]
                    alt = img.get("alt", "")[:40]
                    print(f"  <img src=\"{src}\" alt=\"{alt}\">")
            else:
                print("No .e_richText-11 found")
            break
    else:
        print("未在列表中找到该标题")
