#!/usr/bin/env python3
"""
新乡市生态环境局 - 行政许可与环境影响审批（县区）
https://sthjj.xinxiang.gov.cn/zwgk/public/column/6638717?type=4&catId=6720858&action=list
政府信息公开平台，POST API分页
"""
import re, sys, os, json, time
from datetime import datetime
import requests

SITE_NAME = "xinxiang_xzxk_xq"
BASE_URL = "https://sthjj.xinxiang.gov.cn"
API_LIST = BASE_URL + "/zwgk/site/label/8888"
MAX_PAGES = 20
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

session = requests.Session()
session.headers.update(HEADERS)


def fetch_list(page=1):
    data = {
        "labelName": "publicInfoList",
        "siteId": "6802282",
        "pageSize": "20",
        "pageIndex": str(page),
        "action": "list",
        "isDate": "true",
        "dateFormat": "yyyy-MM-dd",
        "length": "80",
        "organId": "6638717",
        "type": "4",
        "catId": "6720858",
        "cId": "",
        "result": "暂无相关信息",
        "keyWords": "",
        "file": "/xinxiang/xxgk/publicInfoList-xxgk"
    }
    r = session.post(API_LIST, data=data, timeout=30)
    r.encoding = "utf-8"
    return r.text


def parse_list(html):
    """Extract items from list HTML"""
    items = []
    for m in re.finditer(
        r'<a href="(http[^"]+)" class="title[^"]*"[^>]*title="([^"]*)">',
        html
    ):
        href = m.group(1)
        title = m.group(2).strip()
        if not title or not href:
            continue
        items.append({"url": href, "title": title})
    # Match dates in order
    dates = re.findall(r'<span class="date">(\d{4}-\d{2}-\d{2})</span>', html)
    for i, item in enumerate(items):
        item["date"] = dates[i] if i < len(dates) else ""
    return items


def fetch_detail(url):
    html = fetch(url)
    if not html:
        return "", "", ""
    # Title from <title>
    title = ""
    m = re.search(r'<title>([^<]+)', html)
    if m:
        title = re.sub(r'[-_—|].*', '', m.group(1)).strip()
    # Content from ls-article-info
    content = ""
    m = re.search(r'<div class="ls-article-info[^"]*"[^>]*>(.*?)</div>\s*<!--\s*解读、文件', html, re.DOTALL)
    if not m:
        m = re.search(r'<div class="ls-article-info[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if m:
        raw = m.group(1).strip()
        raw = re.sub(r'<script[^>]*>.*?</script>', '', raw, flags=re.DOTALL|re.I)
        raw = re.sub(r'<style[^>]*>.*?</style>', '', raw, flags=re.DOTALL|re.I)
        raw = re.sub(r' style="[^"]*"', '', raw)
        content = raw
    # Date
    date = ""
    m = re.search(r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if m:
        date = m.group(1)
    return title, content, date


def fetch(url):
    try:
        r = session.get(url, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        print(f"  Fetch error: {e}")
        return None


def main():
    total_new = 0
    total_skip = 0
    total_err = 0

    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA busy_timeout=30000")

    for page in range(1, MAX_PAGES + 1):
        print(f"\nPage {page}/{MAX_PAGES}")
        html = fetch_list(page)
        if not html:
            print("  Empty response, done")
            break

        items = parse_list(html)
        if not items:
            print("  No items, done")
            break

        print(f"  Found {len(items)} items")

        for item in items:
            try:
                url = item["url"]
                # Fetch detail
                title, content, date = fetch_detail(url)

                if not content or len(content.strip()) < 50:
                    print(f"  Short: {item['title'][:40]}... ({len(content)} chars)")
                    content = content  # Still save even if short

                if not title:
                    title = item["title"]

                if not date:
                    date = item.get("date", "")

                try:
                    c = conn.execute(
                        """INSERT OR IGNORE INTO gov_raw
                           (title, page_url, source_url, content, publish_date, site_name)
                           VALUES (?, ?, ?, ?, ?, ?)""",
                        (title, url, url, content, date, SITE_NAME)
                    )
                    affected = c.rowcount
                    conn.commit()
                    if affected > 0:
                        total_new += 1
                        print(f"  + {title[:40]}...")
                    else:
                        total_skip += 1
                except Exception as e:
                    print(f"  DB error: {e}")
                    total_err += 1

            except Exception as e:
                print(f"  Error: {e}")
                total_err += 1

        # Check if this is the last page
        if len(items) < 20:
            print("  Last page reached")
            break

    conn.close()
    print(f"\nDone: new={total_new} skip={total_skip} err={total_err}")


if __name__ == "__main__":
    main()
