#!/usr/bin/env python3
"""
连云经济开发区管理委员会 - 开发区文件
URL: http://www.lianyun.gov.cn/lyglyqjjkfqglwyh/zfwj/zfwj.html
CMS: TrueCMS with jpage client-side pagination (all data in #initData div)
Detail: Meta ArticleTitle/PubDate + div#zoom content
"""
import os, re, sys, sqlite3, requests, json, time
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
CUTOFF_DATE = (datetime.now() - timedelta(days=365*3)).strftime("%Y-%m-%d")
BASE = "http://www.lianyun.gov.cn"
LIST_URL = BASE + "/lyglyqjjkfqglwyh/zfwj/zfwj.html"
SITE_NAME = "连云经济开发区-开发区文件"
CATEGORY = "企业环保"

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'
}

def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    conn.row_factory = sqlite3.Row
    return conn

def is_dup(conn, page_url):
    return conn.execute("SELECT id FROM gov_raw WHERE page_url=?", (page_url,)).fetchone() is not None

def insert_item(conn, site_name, title, page_url, publish_date, content, category):
    date_rank = int(publish_date.replace("-", "")) if publish_date and "-" in publish_date else 0
    summary = ""
    if content:
        text_soup = BeautifulSoup(content, 'html.parser')
        plain = text_soup.get_text(strip=True)
        summary = plain[:200]
    try:
        cur = conn.execute(
            "INSERT OR IGNORE INTO gov_raw "
            "(site_name, title, page_url, publish_date, content, date_rank, category, summary) "
            "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
            (site_name, title, page_url, publish_date, content, date_rank, category, summary)
        )
        return cur.rowcount > 0
    except Exception as e:
        print(f"    [ERR] insert: {e}")
        return False

def parse_list_page(html):
    """Extract records from #initData div"""
    items = []
    m = re.search(r'<div id="initData"[^>]*>(.*?)</div>\s*<!--列表内容结束-->', html, re.DOTALL)
    if not m:
        print("[ERR] No initData div found")
        return items
    
    init_html = m.group(1)
    uls = re.findall(r'<ul class="xxgk-list-con">(.*?)</ul>', init_html, re.DOTALL)
    
    for ul in uls:
        # Extract title and URL from <a>
        link_m = re.search(r'<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"', ul)
        if not link_m:
            # Try without title attribute
            link_m = re.search(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', ul, re.DOTALL)
            if link_m:
                href = link_m.group(1)
                title = BeautifulSoup(link_m.group(2), 'html.parser').get_text(strip=True)
            else:
                continue
        else:
            href = link_m.group(1)
            title = link_m.group(2).strip()
        
        # Extract date
        date_m = re.search(r'<span>(\d{4}-\d{2}-\d{2})</span>', ul)
        if not date_m:
            continue
        pub_date = date_m.group(1)
        
        # Build full URL
        if href.startswith("http"):
            full_url = href
        else:
            full_url = BASE + href
        
        items.append((title, full_url, pub_date))
    
    return items

def parse_detail(html):
    """Parse detail page - meta ArticleTitle/PubDate, div#zoom content"""
    title = ""
    content = ""
    pub_date = ""
    
    soup = BeautifulSoup(html, 'html.parser')
    
    # Meta title
    meta = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta and meta.get("content"):
        title = meta["content"].strip()
    
    # Meta pub date
    meta = soup.find("meta", attrs={"name": "PubDate"})
    if meta and meta.get("content"):
        pub_date = meta["content"].strip()
        m2 = re.match(r"(\d{4}-\d{1,2}-\d{1,2})", pub_date)
        if m2:
            pub_date = m2.group(1)
    
    # Content from div#zoom
    zoom = soup.find("div", id="zoom")
    if zoom:
        content = str(zoom)
    
    if not title:
        t_tag = soup.find("title")
        if t_tag:
            title = t_tag.get_text(strip=True)
    
    return title, content, pub_date

def main():
    conn = get_conn()
    
    print(f"Fetching list page: {LIST_URL}")
    resp = requests.get(LIST_URL, headers=HEADERS, timeout=30, verify=False)
    resp.encoding = "utf-8"
    
    items = parse_list_page(resp.text)
    print(f"Found {len(items)} records in list page")
    
    # Filter by cutoff
    active = [(t, u, d) for t, u, d in items if d >= CUTOFF_DATE]
    skipped = len(items) - len(active)
    print(f"Within 3yr cutoff: {len(active)}, older: {skipped}")
    
    # Fetch detail pages
    new_count = dup_count = error_count = 0
    for idx, (title, url, date_str) in enumerate(active, 1):
        try:
            if is_dup(conn, url):
                dup_count += 1
                continue
            
            detail_resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
            detail_resp.encoding = "utf-8"
            detail_title, content, pub_date = parse_detail(detail_resp.text)
            
            if not detail_title:
                detail_title = title
            if not pub_date:
                pub_date = date_str
            
            if insert_item(conn, SITE_NAME, detail_title, url, pub_date, content, CATEGORY):
                new_count += 1
            else:
                dup_count += 1
        
        except Exception as e:
            error_count += 1
            print(f"  ERROR [{idx}] {title[:40]}: {e}")
        
        conn.commit()
        time.sleep(0.3)
        
        if idx % 10 == 0 or idx == len(active):
            print(f"  Progress {idx}/{len(active)}: +{new_count} new, {dup_count} dup, {error_count} err")
    
    conn.close()
    
    print(f"\n{'='*60}")
    print(f"Summary: +{new_count} new, {dup_count} dup, {error_count} err")
    print(json.dumps({
        "site": SITE_NAME,
        "new": new_count,
        "dup": dup_count,
        "err": error_count,
        "total_found": len(active)
    }, ensure_ascii=False))

if __name__ == "__main__":
    main()
