#!/usr/bin/env python3
"""
多氟多新材料股份有限公司 - 公示公告 爬虫
https://www.dfdchem.com/News/5.html
"""
import os
import sys
import json
import sqlite3
import requests
from bs4 import BeautifulSoup
from datetime import datetime
import urllib3
urllib3.disable_warnings()

BASE_URL = "https://www.dfdchem.com"
LIST_URL = "https://www.dfdchem.com/News/5.html"
SITE_NAME = "多氟多新材料"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 5
INCREMENTAL = "--incremental" in sys.argv

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

conn = sqlite3.connect(DB_PATH)
c = conn.cursor()

def fetch_page(url):
    r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
    r.encoding = 'utf-8'
    return r.text

def parse_list(html):
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for item_div in soup.select('.cbox-1.p_loopitem'):
        date_el = item_div.select_one('.e_timeFormat-5.s_title')
        date_str = date_el.get_text(strip=True) if date_el else ''
        link = item_div.select_one('.e_text-6.s_title a')
        if not link:
            continue
        title = link.get_text(strip=True)
        href = link.get('href', '')
        if not title or not href:
            continue
        if not href.startswith('http'):
            href = BASE_URL + href
        items.append({'title': title, 'url': href, 'date': date_str})
    return items

def parse_detail(html, item):
    soup = BeautifulSoup(html, 'html.parser')
    h1 = soup.find('h1')
    detail_title = h1.get_text(strip=True) if h1 else item['title']
    date_el = soup.select_one('.e_timeFormat-8.s_title')
    detail_date = date_el.get_text(strip=True) if date_el else item['date']
    content_html = ''
    content_text = ''
    attachments = []
    rich = soup.select_one('.e_richText-14')
    if rich:
        content_html = str(rich)
        content_text = rich.get_text('\n', strip=True)
        for a in rich.find_all('a', href=True):
            href = a.get('href', '')
            text = a.get_text(strip=True)
            if not text:
                text = href.split('/')[-1].split('?')[0]
            if any(ext in href.lower() for ext in ['.doc', '.docx', '.pdf', '.xls', '.xlsx', '.zip']):
                attachments.append({'text': text, 'url': href if href.startswith('http') else BASE_URL + href})
    # Check for PDF-only content: if content_text is very short (< 20 chars) and attachments exist
    if len(content_text) < 20 and attachments:
        content_text = f"[{detail_title}]({item['url']})\n\n附件：\n" + "\n".join(f"[{a['text']}]({a['url']})" for a in attachments)
    return {
        'title': detail_title,
        'date': detail_date,
        'content_html': content_html,
        'content_text': content_text,
        'attachments': attachments,
    }

def sync_fts(row_id, title):
    try:
        c.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                  (row_id, title, SITE_NAME, ""))
        conn.commit()
    except:
        pass

def insert_article(item):
    page_url = item['url']
    display_url = page_url.replace("http://", "").replace("https://", "")
    summary = item['content_text'][:500] if item['content_text'] else ""

    c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
    row = c.fetchone()
    existing_id = row[0] if row else None
    if existing_id and not INCREMENTAL:
        # In full mode, update existing
        pass
    elif existing_id:
        return False  # In incremental mode, skip existing

    content_to_store = item['content_html'] if item['content_html'] else item['content_text']
    attachments_json = json.dumps(item['attachments'], ensure_ascii=False) if item['attachments'] else ''

    try:
        c.execute("""INSERT OR REPLACE INTO gov_raw
            (page_url, title, content, publish_date, site_name, source_url, summary, attachments)
            VALUES (?, ?, ?, ?, ?, ?, ?, ?)""",
            (page_url, item['title'], content_to_store, item['date'],
             SITE_NAME, display_url, summary, attachments_json))
        row_id = existing_id or c.lastrowid
        if row_id:
            sync_fts(row_id, item['title'])
        return True
    except Exception as e:
        print(f"  [DB ERROR] {e}")
        return False

def main():
    all_items = []
    pages_to_fetch = 1 if INCREMENTAL else MAX_PAGES
    print(f"=== {SITE_NAME}-公示公告 - {'增量' if INCREMENTAL else '全量'}爬取 ===")

    for page in range(1, pages_to_fetch + 1):
        if page == 1:
            url = LIST_URL
        else:
            offset = (page - 1) * 16
            url = f"https://www.dfdchem.com/News/2021778837727387648-{offset}-16.html"
        print(f"Fetching page {page}: {url}")
        try:
            html = fetch_page(url)
        except Exception as e:
            print(f"  -> Error fetching: {e}")
            continue
        items = parse_list(html)
        print(f"  -> {len(items)} items")
        if not items:
            break
        for item in items:
            try:
                detail_html = fetch_page(item['url'])
                detail = parse_detail(detail_html, item)
            except Exception as e:
                print(f"  -> Error detail {item['url']}: {e}")
                continue
            all_items.append({
                'title': detail['title'],
                'url': item['url'],
                'date': detail['date'],
                'content_html': detail['content_html'],
                'content_text': detail['content_text'],
                'attachments': detail['attachments'],
            })

    print(f"\nTotal unique items: {len(all_items)}")
    inserted = 0
    for item in all_items:
        if insert_article(item):
            inserted += 1
    conn.commit()
    conn.close()
    print(f"Inserted {inserted} new records (auto-syncs FTS)")

if __name__ == '__main__':
    main()
