#!/usr/bin/env python3
import os
"""淄博市生态环境局 - 通知公告(395) + 生态要闻(394)"""
import sys, re, json, requests, sqlite3
from datetime import datetime, timedelta

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://epb.zibo.gov.cn/module/web/jpage/dataproxy.jsp"
SECTIONS = {
    "394": {"name": "淄博市生态环境局-生态要闻", "unitid": "54048"},
    "395": {"name": "淄博市生态环境局-通知公告", "unitid": "54048"},
}
BASE_PARAMS = "appid=1&webid=11&path=/&webname=%E6%B7%84%E5%8D%9A%E5%B8%82%E7%94%9F%E6%80%81%E7%8E%AF%E5%A2%83%E5%B1%80&permissiontype=0"

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

def extract_date(text):
    m = re.search(r"(20\d{2}-\d{2}-\d{2})", str(text))
    return m.group(1) if m else None

def is_recent(d, years=3):
    if not d: return False
    try:
        return datetime.strptime(d, "%Y-%m-%d") >= (datetime.now() - timedelta(days=365*years))
    except: return False

def get_content(url):
    """Fetch article content page for full HTML content"""
    try:
        resp = requests.get(url, headers={"User-Agent": "Mozilla/5.0"}, timeout=10)
        if resp.status_code != 200: return ""
        html = resp.text
        # Try common content containers
        for cls in ["TRS_Editor", "article-content", "content", "maintext", "Custom_UnionStyle", "con_text"]:
            m = re.search(r'<div[^>]*class="[^"]*' + cls + r'[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
            if m: return m.group(1).strip()
        # Fallback: get body content
        m = re.search(r'<body[^>]*>(.*?)</body>', html, re.DOTALL)
        if m: return m.group(1).strip()
        return ""
    except:
        return ""

def fetch_section(columnid, unitid, page=1):
    url = f"{BASE_URL}?page={page}&perPage=300&{BASE_PARAMS}&columnid={columnid}&unitid={unitid}"
    try:
        resp = requests.get(url, headers={"User-Agent": "Mozilla/5.0"}, timeout=15)
        if resp.status_code != 200: return None
        import xml.etree.ElementTree as ET
        tree = ET.fromstring(resp.content)
        total = int(tree.findtext("totalrecord", "0"))
        records = tree.findall(".//record")
        return {"total": total, "records": records}
    except Exception as e:
        print(f"  Error: {e}")
        return None

def crawl(incremental=False):
    conn = get_conn()
    cursor = conn.cursor()
    max_pages = 1 if incremental else 1
    total_new = total_skip = total_old = 0

    for cid, sec in SECTIONS.items():
        print(f"\n=== {sec['name']} ===")
        for page in range(1, max_pages + 1):
            print(f"  Page {page}...")
            result = fetch_section(cid, sec["unitid"], page)
            if not result:
                print("  No data, stopping"); break
            print(f"  Total: {result['total']}, Records: {len(result['records'])}")
            if not result["records"]: break

            for record in result["records"]:
                cdata = record.text.strip() if record.text else ""
                # Extract title, url, date
                m_url = re.search(r"href='([^']+)'", cdata)
                m_title = re.search(r"title='([^']+)'", cdata)
                m_date = re.search(r">(\d{4}-\d{2}-\d{2})<", cdata)
                if not m_url or not m_title: continue
                url = "http://epb.zibo.gov.cn" + m_url.group(1)
                title = m_title.group(1).strip()
                date_str = m_date.group(1) if m_date else None
                if date_str and not is_recent(date_str): continue

                cursor.execute("SELECT id FROM gov_raw WHERE source_url=? OR page_url=?", (url, url))
                if cursor.fetchone():
                    continue

                # Get content from article page
                content = get_content(url)
                if not content:
                    total_skip += 1; continue

                summary = re.sub(r"<[^>]+>", "", content)[:200]
                cursor.execute("""
                    INSERT OR IGNORE INTO gov_raw
                    (source_url, page_url, title, summary, content, site_name, publish_date)
                    VALUES (?, ?, ?, ?, ?, ?, ?)
                """, (url, url, title, summary, content, sec["name"], date_str or ""))
                if cursor.rowcount > 0: total_new += 1

            conn.commit()
            if incremental: break
        conn.commit()

    conn.close()
    print(f"\nDone. New: {total_new}, Skipped: {total_skip}, Old: {total_old}")

if __name__ == "__main__":
    crawl(incremental="incremental" in sys.argv)
