#!/usr/bin/env python3
"""
新乡企业-公告公示 爬虫
https://www.0373ds.com/gongshi/
Discuz! 系统, GBK编码
"""
import re
import json
import time
import argparse
import requests
from bs4 import BeautifulSoup

SITE_NAME = "新乡企业-公告公示"
GROUP = "企业"
BASE_URL = "https://www.0373ds.com/gongshi/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
DB_PATH = "/root/search.db"

def fetch(url, max_retries=3):
    for attempt in range(max_retries):
        try:
            r = requests.get(url, headers=HEADERS, timeout=30)
            raw = r.content
            # Page is UTF-8 but contains some mixed encoding bytes in scripts/ads
            # Use replace mode to handle the few bad bytes
            try:
                return raw.decode('utf-8', errors='replace')
            except:
                try:
                    return raw.decode('gbk', errors='replace')
                except:
                    return raw.decode('gb18030', errors='replace')
        except Exception as e:
            print(f"  Error: {e}, retry {attempt+1}")
        time.sleep(2)
    return None

def parse_list_page(html):
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    for a in soup.find_all('a', href=True):
        href = a['href']
        if '/article-' in href and href.endswith('.html'):
            title = a.get_text(strip=True)
            if title and len(title) > 5:
                if not href.startswith('http'):
                    href = requests.compat.urljoin(BASE_URL, href)
                items.append((title, href))
    return items

def parse_detail(html, url):
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from h1.ph or title tag
    title_el = soup.find('h1', class_='ph')
    title = title_el.get_text(strip=True) if title_el else ""
    if not title:
        t = soup.find('title')
        if t:
            title = t.get_text(strip=True)
            # Clean title - remove suffix after '... - '
            title = re.sub(r'\s*\.{3,}.*$', '', title).strip()
    else:
        # Clean truncated titles. This site's h1 is itself truncated by Discuz template
        # with '... ... ' (multiple ellipses). Strip ALL trailing ellipsis groups.
        title = re.sub(r'\s*(?:\.{3,}\s*)+$', '', title).strip()
        # If h1 still truncated, try rebuilding full title from article_content first paragraphs.
        if '...' in title:
            ac = soup.find(id='article_content')
            if ac:
                full_parts = []
                for p in ac.find_all(['p', 'div']):
                    txt = p.get_text(strip=True)
                    if txt and len(txt) > 8:
                        full_parts.append(txt)
                        if len(full_parts) >= 2:
                            break
                cand = ''.join(full_parts)
                if cand and len(cand) > len(title) and '...' not in cand:
                    title = cand.strip()
    
    # Date from p.xg1
    date = ""
    info_el = soup.find('p', class_='xg1')
    if info_el:
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2}\s+\d{1,2}:\d{2})', info_el.get_text())
        if m:
            date = m.group(1)
    
    # Content
    content = ""
    attachments = []
    # Prefer precise article_content container (excludes click-vote table, prev/next nav, share bar).
    vw = soup.find(id='article_content')
    if not vw:
        vw = soup.find(class_='vw')
    if not vw:
        vw = soup.find('div', class_=lambda c: c and 'bm' in c and 'vw' in c)
    
    if vw:
        for tag in vw.find_all(['script', 'style', 'iframe']):
            tag.decompose()
        
        # Collect paragraphs as <p> HTML — do NOT flatten with get_text('\n')
        # (that breaks inline spans into separate lines: 国务院第\n682\n号令).
        paragraphs = []
        for child in vw.children:
            if not hasattr(child, 'name') or child.name is None:
                continue
            cls = child.get('class') or []
            cls_str = ' '.join(cls) if isinstance(cls, list) else str(cls)
            if child.name in ('h1',) and 'ph' in cls_str:
                continue
            if child.name == 'p' and 'xg1' in cls_str:
                continue
            if child.name == 'div' and cls_str in ('h hm', 'hm'):
                continue
            if child.name == 'div' and 's' in cls_str:
                continue
            # Skip click-vote table, share bar, prev/next nav
            if child.name == 'div' and re.search(r'pren|o cl ptm|share|click', cls_str):
                continue
            # Keep tables as HTML (preserve structure, don't flatten)
            if child.name == 'table':
                tbl = str(child)
                if tbl and len(tbl) > 20:
                    paragraphs.append(tbl)
                continue
            # Paragraph text: join inline spans with NO separator so 国务院第682号令 stays intact
            text = child.get_text('', strip=True)
            if not text or len(text) <= 5:
                continue
            # Convert bare cloud-drive URLs inside the paragraph BEFORE wrapping in <p>
            # (outer <p> wrap below; here just make the URL a link)
            text = re.sub(
                r'(https?://(?:pan\.baidu\.com|aliyundrive\.com|lanzou[^\s<"]*|123pan\.com|quark\.cn)[^\s<"]*)',
                lambda m: f'<a href="{m.group(1)}">全文网盘链接</a>',
                text
            )
            # Drop leftover bare ':' / '链接:' / '提取码' fragment lines
            if re.fullmatch(r'[:：][^\n]*', text) or re.fullmatch(r'链接\s*[:：]?', text):
                continue
            text = re.sub(r'^[:：]\s*', '', text)
            paragraphs.append(f'<p>{text}</p>')
        
        content = '\n'.join(paragraphs)
        
        # Attachments: file links + cloud-drive links (baidu pan etc. often carry no file extension)
        for a in vw.find_all('a', href=True):
            href = a['href']
            is_file = re.search(r'\.(doc|docx|pdf|xls|xlsx|xlsm|rar|zip)$', href, re.I)
            is_cloud = re.search(r'(pan\.baidu\.com|aliyundrive|lanzou|123pan|quark)', href, re.I)
            if is_file or is_cloud:
                if not href.startswith('http'):
                    href = requests.compat.urljoin(url, href)
                if is_cloud:
                    attach_title = "全文网盘链接"
                else:
                    attach_title = a.get_text(strip=True) or href.split('/')[-1]
                attachments.append({"title": attach_title, "url": href})
    
    # Post-pass: clean leftover '链接:' / '提取码' fragments anywhere in content
    if content:
        content = re.sub(r'^\s*链接\s*[:：]?\s*$', '', content, flags=re.M)
        content = re.sub(r'^\s*[:：]\s*[^\s<]{0,20}\s*$', '', content, flags=re.M)
        # collapse multiple blank lines
        content = re.sub(r'\n{3,}', '\n\n', content).strip()
    
    # Fallback
    if not content or len(content.strip()) < 20:
        body = soup.find('body')
        if body:
            lines = [l.strip() for l in body.get_text('\n', strip=True).split('\n') if len(l.strip()) > 10]
            content = '\n'.join(lines[:60])
    
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>'
        if attachments:
            content += "\n\n附件：\n" + "\n".join(f"[{a['title']}]({a['url']})" for a in attachments)
    
    return {
        "title": title,
        "publish_date": date,
        "content": content,
        "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        "page_url": url,
    }

def crawl_pages(max_pages):
    all_items = []
    for page in range(1, max_pages + 1):
        if page == 1:
            url = BASE_URL
        else:
            url = f"{BASE_URL}index.php?page={page}"
        
        print(f"  Page {page}: {url}")
        html = fetch(url)
        if not html:
            continue
        
        items = parse_list_page(html)
        if not items:
            print(f"  No items, stopping")
            break
        print(f"  Found {len(items)} items")
        
        for i, (title, item_url) in enumerate(items):
            print(f"    [{i+1}/{len(items)}] {title[:50]}...")
            detail_html = fetch(item_url)
            if not detail_html:
                continue
            detail = parse_detail(detail_html, item_url)
            detail['site_name'] = SITE_NAME
            detail['group'] = GROUP
            all_items.append(detail)
            time.sleep(0.5)
        
        time.sleep(1)
    
    return all_items

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=5000")
    c = conn.cursor()
    
    inserted = 0
    skipped = 0
    for item in items:
        try:
            c.execute("""
                INSERT OR IGNORE INTO gov_raw 
                (site_name, page_url, title, publish_date, summary, content, category, attachments, script_name)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
            """, (
                item['site_name'], item['page_url'], item['title'],
                item.get('publish_date', ''), '',
                item.get('content', ''), item.get('group', ''),
                item.get('attachments', ''), 'crawl_0373ds_v2.py',
            ))
            if c.rowcount > 0:
                inserted += 1
            else:
                skipped += 1
        except Exception as e:
            print(f"    DB Error: {e}")
    
    conn.commit()
    conn.close()
    return inserted, skipped

def main():
    parser = argparse.ArgumentParser(description=f"爬取{SITE_NAME}")
    parser.add_argument('--test', action='store_true', help='测试模式')
    parser.add_argument('--incremental', action='store_true', help='增量模式')
    parser.add_argument('--max-pages', type=int, default=5, help='最大页数')
    args = parser.parse_args()
    
    if args.test or args.incremental:
        end = 1
        mode = "测试" if args.test else "增量"
    else:
        end = args.max_pages
        mode = "全量(前%d页)" % end
    
    print(f"=== {mode}模式: {SITE_NAME} ===")
    items = crawl_pages(end)
    
    if not items:
        print("No items collected")
        return
    
    print(f"\n共获取 {len(items)} 条数据")
    inserted, skipped = save_to_db(items)
    print(f"入库: 新增 {inserted}, 跳过 {skipped}")
    
    print(f"\n=== 前5条预览 ===")
    for item in items[:5]:
        print(f"  [{item['publish_date']}] {item['title'][:60]}")
        if item['attachments']:
            atts = json.loads(item['attachments'])
            if atts:
                print(f"    附件: {', '.join(a['title'] for a in atts[:3])}")

if __name__ == '__main__':
    main()
