#!/usr/bin/env python3
"""
新乡企业-公告公示 爬虫
https://www.0373ds.com/gongshi/
Discuz! 系统, GBK编码
"""
import re
import json
import time
import argparse
import requests
from bs4 import BeautifulSoup

SITE_NAME = "新乡企业-公告公示"
GROUP = "企业"
BASE_URL = "https://www.0373ds.com/gongshi/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
DB_PATH = "/root/search.db"

def fetch(url, max_retries=3):
    for attempt in range(max_retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            raw = r.content
            # Page is UTF-8 but contains some mixed encoding bytes in scripts/ads
            # Use replace mode to handle the few bad bytes
            try:
                return raw.decode('utf-8', errors='replace')
            except:
                try:
                    return raw.decode('gbk', errors='replace')
                except:
                    return raw.decode('gb18030', errors='replace')
        except Exception as e:
            print(f"  Error: {e}, retry {attempt+1}")
        time.sleep(2)
    return None

def parse_list_page(html):
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    for a in soup.find_all('a', href=True):
        href = a['href']
        if '/article-' in href and href.endswith('.html'):
            title = a.get_text(strip=True)
            if title and len(title) > 5:
                if not href.startswith('http'):
                    href = requests.compat.urljoin(BASE_URL, href)
                items.append((title, href))
    return items

def parse_detail(html, url):
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from h1.ph or title tag
    title_el = soup.find('h1', class_='ph')
    title = title_el.get_text(strip=True) if title_el else ""
    if not title:
        t = soup.find('title')
        if t:
            title = t.get_text(strip=True)
            # Clean title - remove suffix after '... - '
            title = re.sub(r'\s*\.{3,}.*$', '', title).strip()
    else:
        # Clean truncated titles from list page
        title = re.sub(r'\s*\.{3,}\s*$', '', title).strip()
    
    # Date from p.xg1
    date = ""
    info_el = soup.find('p', class_='xg1')
    if info_el:
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2}\s+\d{1,2}:\d{2})', info_el.get_text())
        if m:
            date = m.group(1)
    
    # Content
    content = ""
    attachments = []
    vw = soup.find(class_='vw')
    if not vw:
        vw = soup.find('div', class_=lambda c: c and 'bm' in c and 'vw' in c)
    
    if vw:
        for tag in vw.find_all(['script', 'style', 'iframe']):
            tag.decompose()
        
        # Collect paragraphs
        paragraphs = []
        for child in vw.children:
            if not hasattr(child, 'name') or child.name is None:
                continue
            cls = child.get('class') or []
            cls_str = ' '.join(cls) if isinstance(cls, list) else str(cls)
            if child.name in ('h1',) and 'ph' in cls_str:
                continue
            if child.name == 'p' and 'xg1' in cls_str:
                continue
            if child.name == 'div' and cls_str in ('h hm', 'hm'):
                continue
            if child.name == 'div' and 's' in cls_str:
                continue
            
            text = child.get_text('\n', strip=True)
            if text and len(text) > 5:
                paragraphs.append(text)
        
        content = '\n\n'.join(paragraphs)
        
        # Attachments
        for a in vw.find_all('a', href=True):
            href = a['href']
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|xlsm|rar|zip)$', href, re.I):
                if not href.startswith('http'):
                    href = requests.compat.urljoin(url, href)
                attach_title = a.get_text(strip=True) or href.split('/')[-1]
                attachments.append({"title": attach_title, "url": href})
    
    # Fallback
    if not content or len(content.strip()) < 20:
        body = soup.find('body')
        if body:
            lines = [l.strip() for l in body.get_text('\n', strip=True).split('\n') if len(l.strip()) > 10]
            content = '\n'.join(lines[:60])
    
    if len(content.strip()) < 20:
        content = f"[{title}]({url})"
        if attachments:
            content += "\n\n附件：\n" + "\n".join(f"[{a['title']}]({a['url']})" for a in attachments)
    
    return {
        "title": title,
        "publish_date": date,
        "content": content,
        "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        "page_url": url,
    }

def crawl_pages(max_pages):
    all_items = []
    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE_URL
        else:
            url = f"{BASE_URL}index.php?page={page}"
        
        print(f"  Page {page}: {url}")
        html = fetch(url)
        if not html:
            continue
        
        items = parse_list_page(html)
        if not items:
            print(f"  No items, stopping")
            break
        print(f"  Found {len(items)} items")
        
        for i, (title, item_url) in enumerate(items):
            print(f"    [{i+1}/{len(items)}] {title[:50]}...")
            detail_html = fetch(item_url)
            if not detail_html:
                continue
            detail = parse_detail(detail_html, item_url)
            detail['site_name'] = SITE_NAME
            detail['group'] = GROUP
            all_items.append(detail)
            time.sleep(0.5)
        
        time.sleep(1)
    
    return all_items

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    conn.execute("PRAGMA busy_timeout=5000")
    c = conn.cursor()
    
    inserted = 0
    skipped = 0
    for item in items:
        try:
            c.execute("""
                INSERT OR IGNORE INTO gov_raw 
                (site_name, page_url, title, publish_date, summary, content, category, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)
            """, (
                item['site_name'], item['page_url'], item['title'],
                item.get('publish_date', ''), '',
                item.get('content', ''), item.get('group', ''),
                item.get('attachments', ''),
            ))
            if c.rowcount > 0:
                inserted += 1
            else:
                skipped += 1
        except Exception as e:
            print(f"    DB Error: {e}")
    
    conn.commit()
    conn.close()
    return inserted, skipped

def main():
    parser = argparse.ArgumentParser(description=f"爬取{SITE_NAME}")
    parser.add_argument('--test', action='store_true', help='测试模式')
    parser.add_argument('--incremental', action='store_true', help='增量模式')
    parser.add_argument('--max-pages', type=int, default=5, help='最大页数')
    args = parser.parse_args()
    
    if args.test or args.incremental:
        end = 1
        mode = "测试" if args.test else "增量"
    else:
        end = args.max_pages
        mode = "全量(前%d页)" % end
    
    print(f"=== {mode}模式: {SITE_NAME} ===")
    items = crawl_pages(end)
    
    if not items:
        print("No items collected")
        return
    
    print(f"\n共获取 {len(items)} 条数据")
    inserted, skipped = save_to_db(items)
    print(f"入库: 新增 {inserted}, 跳过 {skipped}")
    
    print(f"\n=== 前5条预览 ===")
    for item in items[:5]:
        print(f"  [{item['publish_date']}] {item['title'][:60]}")
        if item['attachments']:
            atts = json.loads(item['attachments'])
            if atts:
                print(f"    附件: {', '.join(a['title'] for a in atts[:3])}")

if __name__ == '__main__':
    main()
