#!/usr/bin/env python3
"""
多氟多新材料股份有限公司 - 公示公告 爬虫
https://www.dfdchem.com/News/5.html
"""
import os
import sys
import json
import sqlite3
import requests
from bs4 import BeautifulSoup
from datetime import datetime
import urllib3
urllib3.disable_warnings()

BASE_URL = "https://www.dfdchem.com"
LIST_URL = "https://www.dfdchem.com/News/5.html"
SITE_NAME = "多氟多新材料"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
MAX_PAGES = 5
INCREMENTAL = "--incremental" in sys.argv

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

conn = sqlite3.connect(DB_PATH, timeout=60)
c = conn.cursor()

# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def fetch_page(url):
    r = requests.get(url, headers=HEADERS, timeout=30, verify=False)
    r.encoding = 'utf-8'
    return r.text

def parse_list(html):
    soup = BeautifulSoup(html, 'html.parser')
    items = []
    for item_div in soup.select('.cbox-1.p_loopitem'):
        date_el = item_div.select_one('.e_timeFormat-5.s_title')
        date_str = date_el.get_text(strip=True) if date_el else ''
        link = item_div.select_one('.e_text-6.s_title a')
        if not link:
            continue
        title = link.get_text(strip=True)
        href = link.get('href', '')
        if not title or not href:
            continue
        if not href.startswith('http'):
            href = BASE_URL + href
        items.append({'title': title, 'url': href, 'date': date_str})
    return items

def parse_detail(html, item):
    soup = BeautifulSoup(html, 'html.parser')
    h1 = soup.find('h1')
    detail_title = h1.get_text(strip=True) if h1 else item['title']
    date_el = soup.select_one('.e_timeFormat-8.s_title')
    detail_date = date_el.get_text(strip=True) if date_el else item['date']
    content_html = ''
    content_text = ''
    attachments = []
    rich = soup.select_one('.e_richText-14')
    if rich:
        content_html = str(rich)
        content_text = body_text(rich)
        for a in rich.find_all('a', href=True):
            href = a.get('href', '')
            text = a.get_text(strip=True)
            if not text:
                text = href.split('/')[-1].split('?')[0]
            if any(ext in href.lower() for ext in ['.doc', '.docx', '.pdf', '.xls', '.xlsx', '.zip']):
                attachments.append({'text': text, 'url': href if href.startswith('http') else BASE_URL + href})
    # Check for PDF-only content: if content_text is very short (< 20 chars) and attachments exist
    if len(content_text) < 20 and attachments:
        content_text = f'<p><a href="{item["url"]}">{detail_title}</a></p>\n\n附件：\n' + "\n".join(f'<p><a href="{a["url"]}">{a["text"]}</a></p>' for a in attachments)
    return {
        'title': detail_title,
        'date': detail_date,
        'content_html': content_html,
        'content_text': content_text,
        'attachments': attachments,
    }

def sync_fts(row_id, title):
    try:
        c.execute("INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary) VALUES (?, ?, ?, ?)",
                  (row_id, title, SITE_NAME, ""))
        conn.commit()
    except:
        pass

def insert_article(item):
    page_url = item['url']
    display_url = page_url.replace("http://", "").replace("https://", "")
    summary = item['content_text'][:500] if item['content_text'] else ""

    c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
    row = c.fetchone()
    existing_id = row[0] if row else None
    if existing_id and not INCREMENTAL:
        # In full mode, update existing
        pass
    elif existing_id:
        return False  # In incremental mode, skip existing

    content_to_store = item['content_html'] if item['content_html'] else item['content_text']
    attachments_json = json.dumps(item['attachments'], ensure_ascii=False) if item['attachments'] else ''

    try:
        c.execute("""INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, source_url, summary, attachments, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, 'crawl_dfdchem.py')""",
            (page_url, item['title'], content_to_store, item['date'],
             SITE_NAME, display_url, summary, attachments_json))
        row_id = existing_id or c.lastrowid
        if row_id:
            sync_fts(row_id, item['title'])
        return True
    except Exception as e:
        print(f"  [DB ERROR] {e}")
        return False

def main():
    all_items = []
    pages_to_fetch = 1 if INCREMENTAL else MAX_PAGES
    print(f"=== {SITE_NAME}-公示公告 - {'增量' if INCREMENTAL else '全量'}爬取 ===")

    for page in range(1, pages_to_fetch + 1):
        if page == 1:
            url = LIST_URL
        else:
            offset = (page - 1) * 16
            url = f"https://www.dfdchem.com/News/2021778837727387648-{offset}-16.html"
        print(f"Fetching page {page}: {url}")
        try:
            html = fetch_page(url)
        except Exception as e:
            print(f"  -> Error fetching: {e}")
            continue
        items = parse_list(html)
        print(f"  -> {len(items)} items")
        if not items:
            break
        for item in items:
            try:
                detail_html = fetch_page(item['url'])
                detail = parse_detail(detail_html, item)
            except Exception as e:
                print(f"  -> Error detail {item['url']}: {e}")
                continue
            all_items.append({
                'title': detail['title'],
                'url': item['url'],
                'date': detail['date'],
                'content_html': detail['content_html'],
                'content_text': detail['content_text'],
                'attachments': detail['attachments'],
            })

    print(f"\nTotal unique items: {len(all_items)}")
    inserted = 0
    for item in all_items:
        if insert_article(item):
            inserted += 1
    conn.commit()
    conn.close()
    print(f"Inserted {inserted} new records (auto-syncs FTS)")

if __name__ == '__main__':
    main()
