#!/usr/bin/env python3
"""泰兴市人民政府-生态环境 爬虫"""
import json, re, sys, os, time
import urllib.request, urllib.parse
import sqlite3

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "taixing.gov.cn-生态环境"
GROUP = "环评公示"
CUTOFF_DATE = "2023-06-16"

API_URL = "https://www.taixing.gov.cn/api-gateway/jpaas-publish-server/front/page/build/unit"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Referer": "https://www.taixing.gov.cn/zwgk/xxgk/zdly/sthj/index.html"
}
BASE_PARAMS = {
    "pageType": "column",
    "tagId": "信息公开内容2",
    "parseType": "bulidstatic",
    "webId": "569012ebb4b9462a9f6c633227a8ef5b",
    "pageId": "NaLBGqC9X7yXPzYhbzhoI",
    "tplSetId": "a62207fcac654afba2ff9d55a7839094"
}

total_inserted = 0
total_skipped = 0

def fetch_list(page):
    params = dict(BASE_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page, "pageSize": 15}, ensure_ascii=False)
    url = API_URL + "?" + urllib.parse.urlencode(params)
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        data = json.loads(resp.read())
        html = data["data"]["html"]
        # Extract links with titles and dates
        pattern = re.compile(
            r'href="(/zwgk/xxgk/zdly/sthj/art/\d+[^"]*?/art_[^"]+\.html)"[^>]*title="([^"]*)"[^>]*>([^<]+)</a>'
            r'.*?<td[^>]*width="14%"[^>]*>(\d{4}-\d{2}-\d{2})<',
            re.DOTALL
        )
        items = pattern.findall(html)
        if not items:
            # Fallback: simpler extraction
            links = re.findall(
                r'href="(/zwgk/xxgk/zdly/sthj/art/[^/]+/art_[^"]+\.html)"[^>]*>([^<]+)</a>',
                html
            )
            dates = re.findall(r'>(\d{4}-\d{2}-\d{2})<', html)
            items = []
            for i, (href, title) in enumerate(links):
                dt = dates[i+1] if i+1 < len(dates) else ""
                items.append((href, title, title, dt))
        return items
    except Exception as e:
        print(f"  Error fetching page {page}: {e}")
        return []

def fetch_detail(url):
    full_url = "https://www.taixing.gov.cn" + url
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode("utf-8", errors="replace")
        # Extract title
        title_m = re.search(r'<p class="con-title">(.*?)</p>', html)
        title = title_m.group(1).strip() if title_m else ""
        # Extract content
        content_m = re.search(r'<div class="main-txt">(.*?)</div>\s*<!--', html, re.DOTALL)
        if not content_m:
            content_m = re.search(r'<div class="main-txt">(.*?)</div>', html, re.DOTALL)
        content = content_m.group(1).strip() if content_m else ""
        # Extract publish date from meta table
        date_m = re.search(r'发文日期[^<]*<td[^>]*>([^<]+)</td>', html)
        if date_m:
            pub_date = date_m.group(1).strip()
        else:
            # From small-title
            date_m2 = re.search(r'发布日期：(\d{4}-\d{2}-\d{2})', html)
            pub_date = date_m2.group(1) if date_m2 else ""
        # Extract source
        source_m = re.search(r'信息来源：([^<]+)', html)
        source = source_m.group(1).strip() if source_m else ""
        # Extract attachments if any
        return {
            "title": title or url.split("/")[-1],
            "content": content,
            "pub_date": pub_date,
            "source": source,
            "page_url": full_url
        }
    except Exception as e:
        print(f"  Error fetching detail {url}: {e}")
        return None

def save_to_db(item):
    global total_inserted, total_skipped
    if item["pub_date"] and item["pub_date"] < CUTOFF_DATE:
        total_skipped += 1
        return
    try:
        db = sqlite3.connect(DB_PATH, timeout=60)
        c = db.cursor()
        # Check existing by page_url
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["page_url"],))
        if c.fetchone():
            total_skipped += 1
            db.close()
            return
        c.execute(
            "INSERT OR IGNORE INTO gov_raw (title, content, publish_date, site_name, source_url, page_url) VALUES (?,?,?,?,?,?)",
            (item["title"], item["content"], item["pub_date"], SITE_NAME, item["page_url"], item["page_url"])
        )
        if c.rowcount > 0:
            inserted_id = c.lastrowid
            # Sync to FTS
            c.execute("INSERT OR IGNORE INTO gov_search(rowid, title, content, summary, site_name, publish_date) VALUES (?,?,?,?,?,?)",
                      (inserted_id, item["title"], item["content"], item["content"][:500] if item["content"] else "", SITE_NAME, item["pub_date"]))
            total_inserted += 1
        db.commit()
        db.close()
    except Exception as e:
        print(f"  DB error: {e}")

def main():
    global total_inserted, total_skipped
    print(f"Starting crawl: {SITE_NAME}")
    print(f"Cutoff date: {CUTOFF_DATE}")

    # First page to get total count
    params = dict(BASE_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": 1, "pageSize": 15}, ensure_ascii=False)
    url = API_URL + "?" + urllib.parse.urlencode(params)
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    data = json.loads(resp.read())
    html = data["data"]["html"]
    count_m = re.search(r'count="(\d+)"', html)
    total_records = int(count_m.group(1)) if count_m else 340
    print(f"Total records: {total_records}")

    for page in range(1, 999):
        items = fetch_list(page)
        if not items:
            print(f"  Page {page}: no items, stopping")
            break
        print(f"  Page {page}: {len(items)} items")
        seen_urls = set()
        for href, _title, text, date in items:
            if href in seen_urls:
                continue
            seen_urls.add(href)
            # Skip old items before fetching detail
            if date and date < CUTOFF_DATE:
                total_skipped += 1
                continue
            detail = fetch_detail(href)
            if detail:
                print(f"    {detail['pub_date']} - {detail['title'][:50]}")
                save_to_db(detail)
            time.sleep(0.5)
        # Stop if last page
        if page * 15 >= total_records:
            break
        time.sleep(1)

    print(f"\nDone! Inserted: {total_inserted}, Skipped: {total_skipped}")

if __name__ == "__main__":
    main()
