#!/usr/bin/env python3
"""
东港市政府 - 生态环境领域 (donggang.gov.cn)
CMS: 自定义 PHP, JS 分页 (client-side only, 30条)
List: /dgszf/zfxxgk/fdzdgknr/zdly/sthjly/glist.html
  Container: ul > li > div.time (date) + div.name > a (title)
Detail: /html/DGSZF/{YYYYMM}/{id}.html
  Content: div.center-info (p paragraphs)
  Title: <title> (strip suffix)
  Date: YYYY-MM-DD from page
  No WAF, curl OK, 30 items on single page (2024-2026)
"""
import sys
import re
import os
import sqlite3
import urllib.request
import ssl

# === CONFIG ===
SITE_NAME = "东港市-生态环境领域"
BASE_URL = "https://www.donggang.gov.cn"
LIST_URL = BASE_URL + "/dgszf/zfxxgk/fdzdgknr/zdly/sthjly/glist.html"
DB_PATH = "/root/search.db"
CUTOFF_DATE = "2023-01-01"

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE


def fetch(url):
    req = urllib.request.Request(url, headers={
        "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    })
    try:
        resp = urllib.request.urlopen(req, timeout=20, context=ssl_ctx)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        print(f"  [FETCH ERROR] {url}: {e}")
        return None


def parse_list(html):
    """Parse list - extract items from ul > li"""
    items = []
    # Find the list ul
    ul_idx = html.find('<ul style="display:block;">')
    if ul_idx < 0:
        return items
    
    ul_end = html.find('</ul>', ul_idx)
    ul_html = html[ul_idx:ul_end + 5] if ul_end > 0 else html[ul_idx:]
    
    pattern = r'<li>.*?<div class="time">(\d{4}-\d{2}-\d{2})</div>.*?<div class="name">.*?<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>'
    for match in re.finditer(pattern, ul_html, re.DOTALL):
        date = match.group(1).strip()
        href = match.group(2).strip()
        title = match.group(3).strip()
        # Strip [生态环境领域] prefix
        title = re.sub(r'^\[.*?\]\s*', '', title)
        if not href.startswith("http"):
            href = BASE_URL + href
        items.append({"url": href, "title": title, "date": date})
    return items


def parse_detail(html):
    """Extract title, date, content"""
    # Title from <title>
    title = ""
    tm = re.search(r'<title>(.*?)</title>', html)
    if tm:
        t = tm.group(1)
        # Strip suffixes: "-生态环境领域-东港市人民政府"
        title = re.sub(r'[-–—][^-–—]*-[^-–—]*$', '', t).strip()
        if not title:
            title = t

    # Date
    date = ""
    dm = re.search(r'(\d{4}-\d{2}-\d{2})', html)
    if dm:
        date = dm.group(1)

    # Content from center-info
    content = ""
    idx = html.find('class="center-info"')
    if idx < 0:
        idx = html.find('class="center-info"')
    if idx > 0:
        start = html.rfind("<div", 0, idx)
        if start < 0: start = idx
        end = html.find('<div class="info-bottom"', idx)
        if end < 0: end = html.find('class="wenjianfeizhi"', idx)
        if end < 0: end = html.find('<div class="fl"', idx)
        if end < 0: end = start + 3000
        
        section = html[start:end]
        text = re.sub(r'<script[^>]*>.*?</script>', '', section, flags=re.DOTALL)
        text = re.sub(r'<style[^>]*>.*?</style>', '', text, flags=re.DOTALL)
        text = re.sub(r'<br\s*/?>', '\n', text)
        text = re.sub(r'</p>', '\n\n', text)
        text = text.replace('\r', '').replace('&nbsp;', ' ')
        
        clean = re.sub(r'<[^>]+>', '', text)
        clean = re.sub(r'[ \t]+', ' ', clean)
        clean = re.sub(r' *\n *', '\n', clean)
        clean = re.sub(r'\n{3,}', '\n\n', clean).strip()
        content = clean

    return {"title": title, "date": date, "content": content}


def insert_to_db(items):
    conn = sqlite3.connect(DB_PATH, timeout=30)
    c = conn.cursor()
    inserted = 0
    skipped = 0
    for item in items:
        title = item.get("title", "")
        url = item.get("url", "")
        date = item.get("date", "")
        content = item.get("content", "")
        if not content and not title:
            skipped += 1
            continue
        try:
            c.execute("""
                INSERT OR IGNORE INTO gov_raw
                (title, page_url, site_name, publish_date, content, summary, attachments, source_url, date_rank)
                VALUES (?, ?, ?, ?, ?, '', '', ?, CAST(strftime('%s', ?) AS INTEGER))
            """, (title, url, SITE_NAME, date, content, url, date))
            if c.rowcount > 0:
                inserted += 1
            else:
                skipped += 1
        except Exception as e:
            print(f"  [DB ERROR] {title[:30]}: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    return inserted, skipped


def main():
    print(f"[INFO] {SITE_NAME} - 爬虫")
    
    html = fetch(LIST_URL)
    if not html:
        print("[ERROR] Cannot fetch list")
        return
    
    items = parse_list(html)
    print(f"  Found {len(items)} items")
    
    all_items = []
    for item in items:
        if item["date"] < CUTOFF_DATE:
            print(f"  [STOP] Date {item['date']} < {CUTOFF_DATE}")
            break
        print(f"    {item['date']} {item['title'][:50]}...")
        detail_html = fetch(item["url"])
        if not detail_html:
            print(f"    [SKIP] Cannot fetch detail")
            continue
        detail = parse_detail(detail_html)
        if detail["title"]: item["title"] = detail["title"]
        if detail["date"]: item["date"] = detail["date"]
        item["content"] = detail["content"]
        all_items.append(item)
    
    if all_items:
        new, old = insert_to_db(all_items)
        print(f"  [DB] +{new} new, {old} existing")
        print(f"\n[DONE] 新增: {new}, 跳过: {old}")
        if new > 0:
            conn = sqlite3.connect(DB_PATH, timeout=30)
            conn.execute("SELECT 1 /* noop: gov_search 由触发器维护, 无需 rebuild */")
            conn.commit()
            conn.close()
            print("FTS rebuilt")


if __name__ == "__main__":
    main()
