#!/usr/bin/env python3
"""
看福清 (fqlook.cn) - 福清企业动态 论坛爬虫
栏目: forum-120 (福清企业动态)
CMS: Discuz! X3.4
详情: 使用可打印版 (action=printable) 免登录看内容
"""
import requests
import re
import json
import time
import sys
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://www.fqlook.cn"
LIST_URL = BASE_URL + "/forum-120-1.html"
PRINTABLE_URL = BASE_URL + "/forum.php?mod=viewthread&action=printable"
SITE_NAME = "看福清-企业动态"
GROUP = "企业"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

def fetch(url):
    r = requests.get(url, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    return r.text

def parse_list(html):
    """解析列表页，返回 [(title, tid, author, date), ...]"""
    items = []
    
    # Find all list items: <li id="normalthread_NNN"> or <li id="stickthread_NNN">
    for li in re.finditer(r'<li[^>]*id="(?:normal|stick)thread_(\d+)"[^>]*>(.*?)</li>', html, re.DOTALL):
        tid = li.group(1)
        li_html = li.group(2)
        
        # Title from <a href="http://www.fqlook.cn/thread-{tid}-...">title</a>
        title = ''
        tm = re.search(r'<a[^>]*href="https?://www\.fqlook\.cn/thread-' + tid + r'[^"]*"[^>]*>(.*?)</a>', li_html, re.DOTALL)
        if tm:
            title = re.sub(r'<[^>]+>', '', tm.group(1)).strip()
        if not title:
            continue
        
        # Author from <div class="zz_aw_meta_author-basic">...<a>author</a>
        author = ''
        am = re.search(r'zz_aw_meta_author-basic[^>]*>.*?<a[^>]*>(.*?)</a>', li_html, re.DOTALL)
        if am:
            author = re.sub(r'<[^>]+>', '', am.group(1)).strip()
        
        # Date from <span>发表于</span><span>YYYY-M-D</span>
        date = ''
        dm = re.search(r'<span>发表于</span>\s*<span>(\d{4}-\d{1,2}-\d{1,2})</span>', li_html)
        if dm:
            date = dm.group(1).strip()
        
        items.append((title, tid, author, date))
    
    return items

def get_total_pages(html):
    """获取总页数"""
    # First try totalpage attribute
    m = re.search(r'totalpage="(\d+)"', html)
    if m:
        return int(m.group(1))
    # Fallback: pg class pagination
    m = re.search(r'class="pg"[^>]*>(.*?)</div>', html, re.DOTALL)
    if m:
        pager = m.group(1)
        pages = re.findall(r'href="forum-120-(\d+)\.html"', pager)
        if pages:
            return max(int(p) for p in pages)
        nums = re.findall(r'<a[^>]>(\d+)</a>', pager)
        if nums:
            ns = [int(n) for n in nums if n.isdigit()]
            if ns:
                return max(ns)
        for span in re.findall(r'<span[^>]*>(.*?)</span>', pager):
            text = re.sub(r'<[^>]+>', '', span).strip()
            m2 = re.search(r'(\d+)\s*页', text)
            if m2:
                return int(m2.group(1))
    return 1

def parse_detail(html, url, tid):
    """解析详情页（可打印版），返回 (title, date, content, author, attachments)"""
    title = ''
    m = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
    if m:
        t = m.group(1).strip()
        if ' - ' in t:
            title = t.split(' - ')[0].strip()
        else:
            title = t
    
    m = re.search(r'<b>\s*标题:\s*</b>\s*(.*?)\s*<b>', html, re.DOTALL)
    if m:
        title2 = re.sub(r'<[^>]+>', '', m.group(1)).strip()
        if title2:
            title = title2
    
    # If Discuz! title has ellipsis, try getting full title from post body
    if title and "..." in title:
        body = re.search(r"<body>(.*?)</body>", html, re.DOTALL)
        if body:
            raw = body.group(1)
            parts = raw.split("标题:")
            for p in parts[2:]:
                lines = p.split("<br")
                for line in lines:
                    text = re.sub(r"<[^>]+>", "", line).strip()
                    text = text.replace("&nbsp;", " ").strip()
                    if text and len(text) > len(title) and "..." not in text:
                        title = text
                        break
                if "..." not in title:
                    break
    
    date = ''
    m = re.search(r'<b>\s*时间:\s*</b>\s*(\d{4}-\d{1,2}-\d{1,2})', html)
    if m:
        date = m.group(1).strip()
    
    if not date:
        m = re.search(r'作者:.*?时间:\s*(\d{4}-\d{1,2}-\d{1,2})', html)
        if m:
            date = m.group(1).strip()
    
    author = ''
    m = re.search(r'<b>\s*作者:\s*</b>\s*(.*?)\s*<b>\s*时间:', html, re.DOTALL)
    if m:
        author = re.sub(r'<[^>]+>', '', m.group(1)).strip()
    
    if not author:
        m = re.search(r'作者:\s*(.*?)\s{2,}时间:', html)
        if m:
            author = m.group(1).strip()
    
    author = author.replace('&nbsp;', ' ').strip()
    author = re.sub(r'\s+', ' ', author).strip()
    
    if date:
        parts = date.split('-')
        if len(parts) == 3:
            date = f"{parts[0]}-{int(parts[1]):02d}-{int(parts[2]):02d}"
    
    content = ''
    attachments = []
    
    td = re.search(r'<td class="t_f"[^>]*>(.*?)</td>', html, re.DOTALL)
    if td:
        raw = td.group(1)
        for a in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', raw):
            href = a.group(1)
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                name = re.sub(r'<[^>]+>', '', a.group(2)).strip() or href.split('/')[-1]
                attachments.append({"url": urljoin(BASE_URL, href), "name": name})
        
        raw = re.sub(r'<div[^>]*class="attach_nopermission[^"]*"[^>]*>.*?</div>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<ignore_js_op[^>]*>.*?</ignore_js_op>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<div[^>]*class="tip[^"]*"[^>]*>.*?</div>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<script[^>]*>.*?</script>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<link[^>]*/>', '', raw)
        
        text = re.sub(r'<br\s*/?>', '\n', raw)
        text = re.sub(r'<[^>]+>', '', text)
        text = text.replace('&nbsp;', ' ').replace('\xa0', ' ')
        text = re.sub(r'\n{3,}', '\n\n', text).strip()
        text = re.sub(r'^\s*[\d,\s]*\n', '', text)
        if text:
            content = text
    else:
        body = re.search(r'<body>(.*?)</body>', html, re.DOTALL)
        if body:
            raw = body.group(1)
            
            parts = raw.split('标题:')
            if len(parts) >= 3:
                actual = '标题:' + parts[2]
            elif len(parts) >= 2:
                actual = '标题:' + parts[1]
            else:
                actual = raw
            
            hr = actual.rfind('<hr noshade')
            if hr >= 0:
                actual = actual[:hr]
            
            actual = re.sub(r'<b>作者:.*?</b>', '', actual)
            actual = re.sub(r'<b>时间:.*?</b>', '', actual)
            actual = re.sub(r'<b>\s*标题:\s*</b>.*?<br\s*/?>', '', actual)
            actual = re.sub(r'<script[^>]*>.*?</script>', '', actual, flags=re.DOTALL)
            actual = re.sub(r'<link[^>]*/>', '', actual)
            actual = re.sub(r'<ignore_js_op[^>]*>.*?</ignore_js_op>', '', actual, flags=re.DOTALL)
            
            for a in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*>(.*?)</a>', actual):
                href = a.group(1)
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                    name = re.sub(r'<[^>]+>', '', a.group(2)).strip() or href.split('/')[-1]
                    attachments.append({"url": urljoin(BASE_URL, href), "name": name})
            
            text = re.sub(r'<br\s*/?>', '\n', actual)
            text = re.sub(r'<[^>]+>', '', text)
            text = text.replace('&nbsp;', ' ').replace('\xa0', ' ')
            text = re.sub(r'\n{3,}', '\n\n', text).strip()
            content = text
    
    content = re.sub(r'[,，、]?\s*下载次数: \d+\s*', '', content)
    content = re.sub(r'^\s*上传\s*', '', content, flags=re.MULTILINE)
    content = re.sub(r'^\s*点击文件名下载附件\s*', '', content, flags=re.MULTILINE)
    content = re.sub(r'\n{3,}', '\n\n', content).strip()
    
    return title, date, content, author, attachments

def crawl_all(max_pages=None):
    """全量爬取"""
    all_items = []
    
    print(f"Fetching page 1: {LIST_URL}")
    html = fetch(LIST_URL)
    items = parse_list(html)
    print(f"  -> {len(items)} items")
    all_items.extend(items)
    
    total_pages = get_total_pages(html)
    print(f"Total pages: {total_pages}")
    
    if max_pages and max_pages > 0:
        total_pages = min(total_pages, max_pages)
    
    for page in range(2, total_pages + 1):
        url = f"{BASE_URL}/forum-120-{page}.html"
        print(f"Fetching page {page}: {url}")
        try:
            html = fetch(url)
            items = parse_list(html)
            print(f"  -> {len(items)} items")
            all_items.extend(items)
            time.sleep(0.5)
        except Exception as e:
            print(f"  ERROR: {e}")
            break
    
    seen = set()
    unique = []
    for item in all_items:
        tid = item[1]
        if tid not in seen:
            seen.add(tid)
            unique.append(item)
    
    print(f"\nTotal unique items: {len(unique)}")
    return unique

def crawl_incremental():
    """增量爬取（仅第1页）"""
    print(f"Fetching page 1 (incremental): {LIST_URL}")
    html = fetch(LIST_URL)
    items = parse_list(html)
    print(f"  -> {len(items)} items")
    return items

def save_to_db(items, db_path="/root/search.db", incremental=False):
    """保存到生产DB"""
    import sqlite3
    
    conn = sqlite3.connect(db_path, timeout=60)
    conn.execute('PRAGMA busy_timeout=5000')
    c = conn.cursor()
    
    new_count = 0
    for i, (title, tid, author, date) in enumerate(items):
        url = f"{PRINTABLE_URL}&tid={tid}"
        try:
            detail_html = fetch(url)
            t, d, content, a, attachments = parse_detail(detail_html, url, tid)
            final_title = t or title
            final_date = d or date
            final_author = a or author
            summary = content[:200] if content else final_title
            attrs_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
            
            c.execute('''INSERT OR IGNORE INTO gov_raw 
                (site_name, source_url, page_url, title, publish_date,
                 content, summary, category, status, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'published', ?)''',
                (SITE_NAME, '', f"https://www.fqlook.cn/thread-{tid}-1-1.html",
                 final_title, final_date, content or '', summary, GROUP, attrs_json))
            if c.rowcount > 0:
                new_count += 1
                if new_count <= 5 or (i+1) % 20 == 0:
                    print(f"  [{i+1}/{len(items)}] {final_title[:40]} +{new_count}")
            
            time.sleep(0.3)
        except Exception as e:
            print(f"  ERROR [tid={tid}]: {e}")
    
    conn.commit()
    conn.close()
    print(f"\nInserted {new_count} new records (trigger auto-syncs FTS)")
    return new_count

if __name__ == '__main__':
    mode = sys.argv[1] if len(sys.argv) > 1 else 'full'
    # Support both --incremental and incremental
    if mode in ('--incremental', 'incremental'):
        mode = 'incremental'
    
    if mode == 'incremental':
        print("=== 看福清-福清企业动态 - 增量爬取 ===")
        items = crawl_incremental()
        save_to_db(items)
    else:
        print("=== 看福清-福清企业动态 - 全量爬取 ===")
        items = crawl_all()
        save_to_db(items)
