#!/usr/bin/env python3
"""
盘锦辽滨沿海经济技术开发区 - 通知公告 爬虫
CMS: P8CMS
栏目: /13775/ - 通知公告
"""
import requests
import re
import json
import time
import sys
import os

BASE_URL = "https://ldwxq.panjin.gov.cn"
LIST_URL = BASE_URL + "/13775/"
SITE_NAME = "盘锦辽滨沿海经济技术开发区-通知公告"
GROUP = "经开区"
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def fetch(url):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    return r.text

def parse_list(html):
    """解析列表页，返回 [(title, url, date), ...]"""
    items = []
    # ul.Q25_ListContlist > li > a[title]
    m = re.search(r'class="Q25_ListContlist"[^>]*>(.*?)</ul>', html, re.DOTALL)
    if not m:
        print("WARN: Q25_ListContlist not found")
        return items
    
    content = m.group(1)
    for a in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>.*?<span>(.*?)</span>\s*<span>(\d{4}-\d{2}-\d{2})</span>', content, re.DOTALL):
        url = a.group(1)
        title = a.group(2).strip()
        title_span = re.sub(r'<[^>]+>', '', a.group(3)).strip()
        date = a.group(4).strip()
        
        # Use title from a[title] attribute (full, no ellipsis)
        final_title = title or title_span
        if not url or not final_title:
            continue
        
        items.append((final_title, url, date))
    
    return items

def get_total_pages(html):
    """获取总页数"""
    m = re.search(r'\d+/(\d+)页', html)
    if m:
        return int(m.group(1))
    m = re.search(r'pages:(\d+)', html)
    if m:
        return int(m.group(1))
    pages = re.findall(r'list-(\d+)\.html', html)
    if pages:
        return max(int(p) for p in pages)
    return 1

def parse_detail(html, url):
    """解析详情页"""
    title = ''
    m = re.search(r'<meta[^>]*ArticleTitle[^>]*content="([^"]*)"', html)
    if m:
        title = m.group(1).strip()
    
    if not title:
        m = re.search(r'<title>(.*?)</title>', html)
        if m:
            t = m.group(1).strip()
            title = t.split('_')[0].strip()
    
    date = ''
    m = re.search(r'<meta[^>]*PubDate[^>]*content="([^"]*)"', html)
    if m:
        date = m.group(1).strip()
    if not date:
        m = re.search(r'发布日期[：:]\s*(\d{4}[-/]\d{1,2}[-/]\d{1,2})', html)
        if m:
            date = m.group(1).strip().replace('/', '-')
    
    content = ''
    attachments = []
    
    m = re.search(r'class="Q25_artcont"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        raw = m.group(1)
        
        # Extract attachment links
        for a in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', raw):
            href = a.group(1)
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                name = re.sub(r'<[^>]+>', '', a.group(2)).strip() or href.split('/')[-1]
                if href.startswith('/'):
                    href = BASE_URL + href
                attachments.append({"url": href, "name": name})
        
        # Convert tables to readable format
        for table in re.finditer(r'<table[^>]*>(.*?)</table>', raw, re.DOTALL):
            table_html = table.group(0)
            rows = re.findall(r'<tr[^>]*>(.*?)</tr>', table_html, re.DOTALL)
            table_text = []
            for row in rows:
                cells = re.findall(r'<t[dh][^>]*>(.*?)</t[dh]>', row, re.DOTALL)
                cell_text = ' | '.join(re.sub(r'<[^>]+>', '', c).strip() for c in cells)
                table_text.append(cell_text)
            raw = raw.replace(table_html, '|\n' + '\n'.join(table_text) + '\n|')
        
        # Clean HTML
        raw = re.sub(r'<script[^>]*>.*?</script>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<style[^>]*>.*?</style>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<br\s*/?>', '\n', raw)
        raw = re.sub(r'<p[^>]*>', '', raw)
        raw = re.sub(r'</p>', '\n\n', raw)
        raw = re.sub(r'<div[^>]*>', '', raw)
        raw = re.sub(r'</div>', '\n', raw)
        raw = re.sub(r'<[^>]+>', '', raw)
        raw = raw.replace('&nbsp;', ' ').replace('&amp;', '&').replace('&lt;', '<').replace('&gt;', '>')
        raw = re.sub(r'[ \t]+', ' ', raw)
        raw = re.sub(r'\n{3,}', '\n\n', raw).strip()
        
        if raw:
            content = raw
    
    # Handle empty content (images only) - use title + URL as fallback
    if not content or len(content.strip()) < 20:
        content = '<p><a href="%s">%s</a></p>' % (url, title)
        # Check for PDF attachments
        pdf_links = re.findall(r'href="([^"]*\.pdf)"', html, re.I)
        if pdf_links:
            content += "\n\n附件PDF链接:\n"
            for pl in pdf_links[:5]:
                if pl.startswith('/'):
                    pl = BASE_URL + pl
                content += '<p><a href="%s">PDF文档</a></p>\n' % pl
    
    return title, date, content, attachments

def crawl_all(max_pages=None):
    """全量爬取"""
    all_items = []
    
    print("Fetching page 1: %s" % LIST_URL)
    html = fetch(LIST_URL)
    items = parse_list(html)
    print("  -> %d items" % len(items))
    all_items.extend(items)
    
    total_pages = get_total_pages(html)
    print("Total pages: %d" % total_pages)
    
    if max_pages and max_pages > 0:
        total_pages = min(total_pages, max_pages)
    
    for page in range(2, total_pages + 1):
        url = "%s/13775/list-%d.html" % (BASE_URL, page)
        print("Fetching page %d: %s" % (page, url))
        try:
            html = fetch(url)
            items = parse_list(html)
            print("  -> %d items" % len(items))
            all_items.extend(items)
            time.sleep(0.3)
        except Exception as e:
            print("  ERROR: %s" % e)
            break
    
    # Deduplicate by URL
    seen = set()
    unique = []
    for item in all_items:
        item_url = item[1]
        if item_url not in seen:
            seen.add(item_url)
            unique.append(item)
    
    print("\nTotal unique items: %d" % len(unique))
    return unique

def crawl_incremental():
    """增量爬取（仅第1页）"""
    print("Fetching page 1 (incremental): %s" % LIST_URL)
    html = fetch(LIST_URL)
    items = parse_list(html)
    print("  -> %d items" % len(items))
    return items

def save_to_db(items):
    """保存到生产DB，含重试"""
    import sqlite3
    import time as time_mod
    
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute('PRAGMA busy_timeout=30000')
    c = conn.cursor()
    
    new_count = 0
    for i, (title, item_url, date) in enumerate(items):
        for attempt in range(3):
            try:
                detail_html = fetch(item_url)
                t, d, content, attachments = parse_detail(detail_html, item_url)
                final_title = t or title
                final_date = d or date
                summary = content[:200] if content else final_title
                attrs_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
                
                c.execute('''INSERT OR IGNORE INTO gov_raw 
                    (site_name, source_url, page_url, title, publish_date,
                     content, summary, category, status, attachments)
                    VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'published', ?)''',
                    (SITE_NAME, '', item_url,
                     final_title, final_date, content or '', summary, GROUP, attrs_json))
                if c.rowcount > 0:
                    new_count += 1
                    if new_count <= 5 or (i+1) % 20 == 0:
                        print("  [%d/%d] %s +%d" % (i+1, len(items), final_title[:40], new_count))
                break
            except sqlite3.OperationalError as e:
                if "locked" in str(e) and attempt < 2:
                    time_mod.sleep(5)
                    continue
                print("  DB ERROR [%s]: %s" % (item_url, e))
                break
            except Exception as e:
                if attempt < 2:
                    time_mod.sleep(3)
                    continue
                print("  ERROR [%s]: %s" % (item_url, e))
                break
        
        time_mod.sleep(0.3)
    
    conn.commit()
    conn.close()
    print("\nInserted %d new records (trigger auto-syncs FTS)" % new_count)
    return new_count

if __name__ == '__main__':
    mode = sys.argv[1] if len(sys.argv) > 1 else 'full'
    
    if mode == 'incremental':
        print("=== 盘锦辽滨-通知公告 - 增量爬取 ===")
        items = crawl_incremental()
        save_to_db(items)
    else:
        print("=== 盘锦辽滨-通知公告 - 全量爬取 ===")
        items = crawl_all()
        save_to_db(items)
