#!/usr/bin/env python3
"""
南丰县人民政府 - 环境保护（环评审批/环境监测/环保公告）
https://www.jxnf.gov.cn/col/col27107/index.html?number=C60000C60004C60001
Hanweb CMS - search.jsp API分页
"""
import sys, os, re, json, time
import requests
import sqlite3
from datetime import datetime

SITE_NAME = "jxnf_hjbh"
LIST_URL = "https://www.jxnf.gov.cn/module/xxgk/search.jsp"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
           "Content-Type": "application/x-www-form-urlencoded"}
MAX_PAGES = 40
POST_DATA_TMPL = (
    "infotypeId=C60000C60004C60001&jdid=4&area=&divid=div7299"
    "&vc_title=&vc_number=&currpage={page}&vc_filenumber=&vc_all=&texttype=&fbtime="
)

def fetch_list(page):
    r = requests.post(LIST_URL, data=POST_DATA_TMPL.format(page=page), headers=HEADERS, timeout=30)
    r.encoding = "utf-8"
    return r.text


def parse_list(html):
    items = []
    for m in re.finditer(
        r'''<a href=['"](http[^'"]+art_27107_\d+\.html)['"][^>]*title=['"]([^'"]*)['"]>''',
        html
    ):
        href = m.group(1)
        title = m.group(2).strip()
        items.append({"url": href, "title": title})
    dates = re.findall(r'<b>\s*(\d{4}-\d{2}-\d{2})', html)
    for i, item in enumerate(items):
        item["date"] = dates[i] if i < len(dates) else ""
    return items


def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        html = r.text
    except Exception as e:
        print(f"  Fetch detail error: {e}")
        return "", "", ""

    # Title from <title>
    title = ""
    m = re.search(r'<title>([^<]+)', html)
    if m:
        title = re.sub(r'[-_—|].*', '', m.group(1)).strip()

    # Content from id=zoom
    content = ""
    m = re.search(r'<div id="zoom"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if m:
        raw = m.group(1).strip()
        raw = re.sub(r'<script[^>]*>.*?</script>', '', raw, flags=re.DOTALL|re.I)
        raw = re.sub(r'<style[^>]*>.*?</style>', '', raw, flags=re.DOTALL|re.I)
        raw = re.sub(r' style="[^"]*"', '', raw)
        content = raw

    # Date
    date = ""
    m = re.search(r'(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}', html)
    if not m:
        m = re.search(r'发布时间[：:]\s*(\d{4}[-年]\d{2}[-月]\d{2})', html)
    if m:
        date = m.group(1).replace("年", "-").replace("月", "-")

    return title, content, date


def main():
    total_new = 0
    total_skip = 0
    total_err = 0

    conn = sqlite3.connect(SEARCH_DB)
    conn.execute("PRAGMA busy_timeout=30000")

    for page in range(1, MAX_PAGES + 1):
        print(f"Page {page}")
        html = fetch_list(page)
        if not html:
            print("  Empty response")
            break

        items = parse_list(html)
        if not items:
            print("  No items")
            break

        # Find total count for pages
        m = re.search(r'nTotalCount[^>]+>(\d+)', html)
        total = m.group(1) if m else "?"
        print(f"  {len(items)} items (total: {total})")

        for item in items:
            try:
                url = item["url"]
                title, content, date = fetch_detail(url)

                if not title:
                    title = item["title"]
                if not date:
                    date = item.get("date", "")

                try:
                    c = conn.execute(
                        """INSERT OR IGNORE INTO gov_raw
                           (title, page_url, source_url, content, publish_date, site_name)
                           VALUES (?, ?, ?, ?, ?, ?)""",
                        (title, url, url, content, date, SITE_NAME)
                    )
                    conn.commit()
                    if c.rowcount > 0:
                        total_new += 1
                except Exception as e:
                    total_err += 1

            except Exception as e:
                print(f"  Error: {e}")
                total_err += 1

        if len(items) < 18:
            print("  Last page")
            break

        time.sleep(0.5)

    print(f"\nDone: new={total_new} skip={len(items)*page-total_new} err={total_err}")
    conn.close()


if __name__ == "__main__":
    main()
