#!/usr/bin/env python3
"""


新余高新区政府门户网站 - 通知公告 爬虫
URL: https://xyhdz.xinyu.gov.cn/C4370/gongsgao/gxqlist.shtml
CMS: 自定义系统
"""
import re, urllib.request, sys, os, time
from urllib.parse import urljoin

import sys as _SYS
_MAX_PG = int(_SYS.argv[1]) if len(_SYS.argv) > 1 and _SYS.argv[1].isdigit() else None
if _MAX_PG is not None:
    print('[AutoPg] max_pages=' + str(_MAX_PG))
# END AUTO PAGES
BASE_URL = "https://xyhdz.xinyu.gov.cn"
LIST_PATH = "/C4370/gongsgao/gxqlist.shtml"
SITE_NAME = "新余高新区政府-通知公告"
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

def ensure_db():
    import sqlite3
    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("""
        CREATE TABLE IF NOT EXISTS gov_raw (
            id INTEGER PRIMARY KEY,
            title TEXT,
            content TEXT,
            publish_date TEXT,
            page_url TEXT UNIQUE,
            source_url TEXT,
            site_name TEXT DEFAULT '',
            category TEXT DEFAULT ''
        )
    """)
    conn.commit()
    return conn

def save_item(conn, title, content, publish_date, url):
    conn.execute(
        "INSERT OR IGNORE INTO gov_raw (title, content, publish_date, page_url, source_url, site_name) VALUES (?, ?, ?, ?, ?, ?)",
        (title, content, publish_date, url, BASE_URL, SITE_NAME)
    )
    conn.commit()

def fetch_page(page_no):
    """获取列表页"""
    if page_no == 1:
        url = BASE_URL + LIST_PATH
    else:
        url = f"{BASE_URL}/C4370/gongsgao/gxqlist_{page_no}.shtml"
    
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    html = resp.read().decode('utf-8', errors='replace')
    
    # 提取列表项
    items = []
    pattern = re.compile(
        r'<a href="([^"]+)"[^>]*>([^<]+)</a>\s*<span>(\d{4}-\d{2}-\d{2})</span>'
    )
    for m in pattern.finditer(html):
        href = m.group(1)
        title = m.group(2).strip()
        date = m.group(3)
        full_url = urljoin(BASE_URL, href)
        items.append((title, date, full_url))
    
    # 获取分页信息
    pg_m = re.search(r"createPageHTML\('page_div',\s*(\d+)", html)
    total_pages = int(pg_m.group(1)) if pg_m else 0
    
    return items, total_pages

def fetch_detail(url):
    """获取详情页"""
    req = urllib.request.Request(url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30)
    html = resp.read().decode('utf-8', errors='replace')
    
    # 标题 - <ucaptitle> 标签
    title_m = re.search(r'<ucaptitle>(.*?)</ucaptitle>', html, re.DOTALL)
    title = title_m.group(1).strip() if title_m else ''
    
    # 日期 - <PUBLISHTIME> 标签
    date_m = re.search(r'<PUBLISHTIME>(.*?)</PUBLISHTIME>', html, re.DOTALL)
    date = date_m.group(1).strip()[:10] if date_m else ''
    
    # 正文 - <div id="zoomcon">
    content_m = re.search(r'<div[^>]*id="zoomcon"[^>]*>(.*?)</div>', html, re.DOTALL)
    if content_m:
        content_html = content_m.group(1)
        # 移除 UCAPCONTENT 外层标签
        content_html = re.sub(r'</?UCAPCONTENT[^>]*>', '', content_html)
        # 图片 base64 嵌入（跳过，正文HTML已有src引用）
        return title, date, content_html.strip()
    
    return title, date, ''

def main():
    conn = ensure_db()
    
    # 先获取第一页，知道总页数
    first_items, total_pages = fetch_page(1)
    print(f"Total pages: {total_pages}")
    
    all_items = []
    for pn in range(1, min(total_pages, _MAX_PG or total_pages)+1):
        if pn == 1:
            items = first_items
        else:
            items, _ = fetch_page(pn)
        print(f"Page {pn}: {len(items)} items")
        all_items.extend(items)
        time.sleep(0.3)
    
    print(f"\nTotal items from list: {len(all_items)}")
    
    for idx, (title, date, url) in enumerate(all_items, 1):
        try:
            det_title, det_date, content = fetch_detail(url)
            use_title = title or det_title
            use_date = date or det_date
            
            text = re.sub(r'<[^>]+>', '', content).strip() if content else ''
            content_ok = len(text) >= 50
            save_item(conn, use_title, content, use_date, url)
            status = "OK" if content_ok else "SHORT"
            if idx <= 5 or content_ok:
                print(f"  [{idx}/{len(all_items)}] {status} {use_date} {use_title[:60]}")
        except Exception as e:
            print(f"  [{idx}/{len(all_items)}] ERROR {url}: {e}")
        
        time.sleep(0.3)
    
    cur = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=? AND content IS NOT NULL AND length(content)>50", (SITE_NAME,))
    ok_count = cur.fetchone()[0]
    total_count = conn.execute("SELECT COUNT(*) FROM gov_raw WHERE site_name=?", (SITE_NAME,)).fetchone()[0]
    
    print(f"\n{'='*50}")
    print(f"完成！总计: {total_count} 条, 有内容: {ok_count} 条")
    
    conn.close()

if __name__ == '__main__':
    main()
