#!/usr/bin/env python3
"""
新乡企业-公告公示 爬虫
https://www.0373ds.com/gongshi/
Discuz! 系统, 页面有混合编码问题 - 使用原始字节提取
"""
import re
import json
import time
import argparse
import requests
from bs4 import BeautifulSoup

SITE_NAME = "新乡企业-公告公示"
GROUP = "企业"
BASE_URL = "https://www.0373ds.com/gongshi/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}
DB_PATH = "/root/search.db"

def fetch_raw(url, max_retries=3):
    """Fetch raw bytes - don't decode the page (mixed encoding)"""
    for attempt in range(max_retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            if r.status_code == 200:
                return r.content
            print(f"  HTTP {r.status_code}, retry {attempt+1}")
        except Exception as e:
            print(f"  Error: {e}, retry {attempt+1}")
        time.sleep(2)
    return None

def safe_decode(data):
    """Decode bytes that are known to be UTF-8, with fallback"""
    if isinstance(data, bytes):
        try:
            return data.decode('utf-8')
        except:
            return data.decode('utf-8', errors='replace')
    return str(data)

def parse_list_page(raw):
    """Parse list page from raw bytes using regex"""
    items = []
    # Find article links in raw bytes
    for m in re.finditer(rb'href="([^"]*/article-\d+[^"]*)"[^>]*>(.*?)</a>', raw, re.DOTALL):
        href = m.group(1).decode('utf-8', errors='replace')
        title_bytes = m.group(2)
        # Clean title bytes - remove any embedded tags
        title_bytes = re.sub(rb'<[^>]+>', b'', title_bytes)
        title = title_bytes.decode('utf-8', errors='replace').strip()
        if title and len(title) > 5:
            if not href.startswith('http'):
                href = requests.compat.urljoin(BASE_URL, href)
            # Remove trailing ...
            title = re.sub(r'\s*\.{3,}\s*$', '', title).strip()
            items.append((title, href))
    return items

def parse_detail(raw, url):
    """Parse detail page from raw bytes"""
    text = raw.decode('utf-8', errors='replace')
    soup = BeautifulSoup(text, 'html.parser')
    
    # Title
    title = ""
    h1 = soup.find('h1', class_='ph')
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        title_match = re.search(rb'<h1[^>]*class="ph"[^>]*>(.*?)</h1>', raw, re.DOTALL)
        if title_match:
            title = safe_decode(title_match.group(1))
    if not title:
        t = soup.find('title')
        if t:
            raw_title = t.get_text(strip=True)
            title = re.sub(r'\s*\.{3,}\s*-?\s*.*$', '', raw_title).strip()
    
    # Clean title
    title = re.sub(r'\s*\.{2,}\s*$', '', title).strip()
    
    # Date
    date = ""
    info = soup.find('p', class_='xg1')
    if info:
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2}\s+\d{1,2}:\d{2})', info.get_text())
        if m:
            date = m.group(1)
    if not date:
        dm = re.search(rb'(\d{4}-\d{1,2}-\d{1,2}\s+\d{1,2}:\d{2})', raw)
        if dm:
            date = dm.group(1).decode()
    
    # Content - extract from the article body in raw bytes
    content = ""
    attachments = []
    
    # Find the vw area
    vw_match = re.search(rb'class="bm vw"[^>]*>(.*?)</div>\s*</div>\s*</div>', raw, re.DOTALL)
    if vw_match:
        vw_bytes = vw_match.group(1)
        # Remove script/style blocks
        vw_bytes = re.sub(rb'<script[^>]*>.*?</script>', b'', vw_bytes, flags=re.DOTALL)
        vw_bytes = re.sub(rb'<style[^>]*>.*?</style>', b'', vw_bytes, flags=re.DOTALL)
        
        # Find paragraph text (skip header/summary area)
        paras = []
        for pm in re.finditer(rb'<p[^>]*>(.*?)</p>', vw_bytes, re.DOTALL):
            p_text = re.sub(rb'<[^>]+>', b'', pm.group(1)).strip()
            p_text = p_text.decode('utf-8', errors='replace').strip()
            if len(p_text) > 5:
                paras.append(p_text)
        
        if paras:
            content = '\n\n'.join(paras)
        
        # Find attachments
        for am in re.finditer(rb'<a[^>]*href="([^"]*\.(?:doc|docx|pdf|xls|xlsx|xlsm|rar|zip)[^"]*)"[^>]*>(.*?)</a>', raw, re.DOTALL):
            ahref = safe_decode(am.group(1))
            atitle = safe_decode(am.group(2))
            if not ahref.startswith('http'):
                ahref = requests.compat.urljoin(url, ahref)
            attachments.append({"title": atitle, "url": ahref})
    
    # Fallback
    if not content or len(content.strip()) < 20:
        lines = [l.strip() for l in text.split('\n') if len(l.strip()) > 10]
        content = '\n'.join(lines[:60])
    
    if len(content.strip()) < 20:
        content = f"[{title}]({url})"
        if attachments:
            content += "\n\n附件：\n" + "\n".join(f"[{a['title']}]({a['url']})" for a in attachments)
    
    return {
        "title": title,
        "publish_date": date,
        "content": content,
        "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        "page_url": url,
    }

def crawl_pages(max_pages):
    all_items = []
    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE_URL
        else:
            url = f"{BASE_URL}index.php?page={page}"
        
        print(f"  Page {page}: {url}")
        raw = fetch_raw(url)
        if not raw:
            continue
        
        items = parse_list_page(raw)
        if not items:
            print(f"  No items, stopping")
            break
        print(f"  Found {len(items)} items")
        
        for i, (title, item_url) in enumerate(items):
            print(f"    [{i+1}/{len(items)}] {title[:50]}...")
            detail_raw = fetch_raw(item_url)
            if not detail_raw:
                continue
            detail = parse_detail(detail_raw, item_url)
            detail['site_name'] = SITE_NAME
            detail['group'] = GROUP
            all_items.append(detail)
            time.sleep(0.5)
        
        time.sleep(1)
    
    return all_items

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA busy_timeout=5000")
    c = conn.cursor()
    inserted = 0; skipped = 0
    for item in items:
        try:
            c.execute("""
                INSERT OR IGNORE INTO gov_raw 
                (site_name, page_url, title, publish_date, summary, content, category, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (item['site_name'], item['page_url'], item['title'],
                  item.get('publish_date',''), '', item.get('content',''),
                  item.get('group',''), item.get('attachments','')))
            if c.rowcount > 0: inserted += 1
            else: skipped += 1
        except Exception as e:
            print(f"    DB Error: {e}")
    conn.commit(); conn.close()
    return inserted, skipped

def main():
    parser = argparse.ArgumentParser(description=f"爬取{SITE_NAME}")
    parser.add_argument('--test', action='store_true')
    parser.add_argument('--incremental', action='store_true')
    parser.add_argument('--max-pages', type=int, default=5)
    args = parser.parse_args()
    
    end = 1 if (args.test or args.incremental) else args.max_pages
    mode = "测试" if args.test else ("增量" if args.incremental else "全量(前%d页)" % end)
    
    print(f"=== {mode}模式: {SITE_NAME} ===")
    items = crawl_pages(end)
    if not items:
        print("No items collected"); return
    
    print(f"\n共获取 {len(items)} 条数据")
    inserted, skipped = save_to_db(items)
    print(f"入库: 新增 {inserted}, 跳过 {skipped}")
    
    print(f"\n=== 前3条预览 ===")
    for item in items[:3]:
        print(f"  [{item['publish_date']}] {item['title'][:60]}")

if __name__ == '__main__':
    main()
