"""调试 - 检查Scrapy响应内容"""
import scrapy


class DebugSpider(scrapy.Spider):
    name = "debug"
    start_urls = [
        "https://www.ahhuoshan.gov.cn/public/column/6597441?type=4&action=list",
    ]
    custom_settings = {
        'ROBOTSTXT_OBEY': False,
        'DOWNLOAD_TIMEOUT': 30,
        'LOG_LEVEL': 'INFO',
        'CONCURRENT_REQUESTS': 1,
    }

    def parse(self, response):
        self.logger.info(f"URL: {response.url}")
        self.logger.info(f"Status: {response.status}")
        self.logger.info(f"Encoding: {response.encoding}")
        self.logger.info(f"Headers: {dict(response.headers)}")
        self.logger.info(f"Body length: {len(response.body)}")

        # Save the raw HTML for debugging
        with open('/tmp/debug_page.html', 'wb') as f:
            f.write(response.body)
        self.logger.info("Saved to /tmp/debug_page.html")

        # XPath tests
        tests = [
            '//li[.//a]',
            '//div[contains(@class, "list") or contains(@class, "news")]//li[.//a]',
            '//div[contains(@class, "content")]',
            '//div[contains(@class, "article")]',
            '//div[@id="content"]',
            '//span[contains(@class, "date")]',
        ]
        for sel in tests:
            nodes = response.xpath(sel)
            self.logger.info(f"XPath '{sel}': {len(nodes)} nodes")
            if nodes:
                for n in nodes[:2]:
                    text = n.xpath('.//text()').getall()
                    clean = ' '.join(t.strip() for t in text if t.strip())[:100]
                    self.logger.info(f"  sample: {clean}")
