#!/usr/bin/env python3
"""四川泸天化股份有限公司 — 安全环保公示
http://www.sclth.com/news/anquan.html
大浪科技 CMS, /news/anquan_{n}.html 分页
"""
import os, sys, re, time
import requests
from bs4 import BeautifulSoup
from datetime import datetime, timedelta
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.sclth.com"
LIST_PATH = "/news/anquan.html"
SITE_NAME = "四川泸天化股份-安全环保"
THRESHOLD = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")
MAX_PAGES = 10

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
session = requests.Session()
session.headers.update(HEADERS)

import sqlite3


def store_item(title, date_str, content, page_url):
    try:
        conn = sqlite3.connect(DB_PATH, timeout=10)
        c = conn.cursor()
        summary = (content.strip()[:200] if content.strip() else '')
        date_rank = 0
        if date_str:
            try:
                date_rank = int(datetime.strptime(date_str[:10], "%Y-%m-%d").timestamp())
            except:
                pass
        c.execute("""INSERT OR IGNORE INTO gov_raw (title, publish_date, content, page_url, source_url, site_name, summary, date_rank)
                     VALUES (?,?,?,?,?,?,?,?)""",
                  (title.strip(), date_str, content.strip(), page_url.strip(), page_url.strip(), SITE_NAME, summary, date_rank))
        a = c.rowcount
        conn.commit()
        conn.close()
        return a
    except Exception as e:
        print(f"[ERROR] 写入失败: {e}")
        return 0


def get_page_url(page):
    if page == 1:
        return BASE_URL + LIST_PATH
    return f"{BASE_URL}/news/anquan_{page}.html"


def parse_list_page(html):
    """解析列表页，返回 [(title, date_str, detail_url), ...]"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.select("ul.list_news li"):
        a = li.find("a")
        if not a:
            continue
        href = a.get("href", "")
        if not href or "articles/" not in href:
            continue
        full_url = href if href.startswith("http") else urljoin(BASE_URL, href)

        h3 = li.find("h3")
        title = h3.get_text(strip=True) if h3 else ""

        date_p = li.find("p", class_="date")
        date_str = date_p.get_text(strip=True) if date_p else ""

        if not title or len(title) < 5:
            continue

        items.append((title, date_str, full_url))

    return items


def extract_attachments(article_soup, page_url):
    """从article中提取PDF/DOC附件链接"""
    attachments = []
    for a in article_soup.find_all("a", href=re.compile(r"\.(pdf|doc|docx)$", re.I)):
        href = a.get("href", "")
        if not href:
            continue
        full_url = href if href.startswith("http") else urljoin(BASE_URL, href)
        text = a.get_text(strip=True) or os.path.basename(href)
        attachments.append((text, full_url))
    return attachments


def parse_detail_page(html, page_url):
    """解析详情页，返回 (title, date_str, content, attachments)"""
    soup = BeautifulSoup(html, "html.parser")

    # Title from <h1>
    title = ""
    h1 = soup.find("h1")
    if h1:
        title = h1.get_text(strip=True)
    if not title:
        tt = soup.find("title")
        if tt:
            title = tt.get_text(strip=True).replace("_安全环保_四川泸天化股份有限公司", "").replace("_四川泸天化股份有限公司", "")

    # Date from the info list
    date_str = ""
    info_ul = soup.select_one("ul:has(li:-soup-contains('时间：'))")
    if info_ul:
        for li in info_ul.find_all("li"):
            m = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", li.get_text())
            if m:
                date_str = m.group(1)
                break
    if not date_str:
        # fallback: find date in first 2000 chars
        for m in re.finditer(r"(\d{4}-\d{1,2}-\d{1,2})", html[:3000]):
            date_str = m.group(1)
            break

    # Content from <article>
    body_parts = []
    attachments = []

    article = soup.find("article")
    if article:
        # Extract text from <p> tags
        for p in article.find_all("p"):
            # Skip empty paragraphs and those containing only attachment links
            t = p.get_text(strip=True)
            if not t:
                continue
            # Check if paragraph only has a single file link
            links = p.find_all("a")
            if links and all(l.get("href", "").lower().endswith((".pdf", ".doc", ".docx")) for l in links):
                # This paragraph contains attachment links - handle attachments separately
                continue
            body_parts.append(t)

        # Extract tables
        for table in article.find_all("table"):
            rows = []
            for tr in table.find_all("tr"):
                cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                if cells:
                    rows.append(" | ".join(cells))
            if rows:
                body_parts.append("\n".join(rows))

        # Extract attachments
        attachments = extract_attachments(article, page_url)
    else:
        # Fallback: just get text from the main content area
        content_div = soup.select_one(".main_all .main_right") or soup.select_one(".main_all > .generic")
        if content_div:
            for p in content_div.find_all("p"):
                t = p.get_text(strip=True)
                if t and len(t) > 3:
                    body_parts.append(t)

    # Build content with attachments
    content = "\n\n".join(body_parts)
    for name, url in attachments:
        content += f'\n\n📎 <p><a href="{url}">{name}</a></p>'

    return title, date_str, content, attachments


def main():
    total_new = 0
    total_skip = 0
    total_fail = 0
    stopped_early = False

    for page in range(1, MAX_PAGES + 1):
        page_url = get_page_url(page)
        print(f"\n=== 第{page}页: {page_url}")
        try:
            r = session.get(page_url, timeout=20)
            r.encoding = "utf-8"
            html = r.text
        except Exception as e:
            print(f"[ERROR] 获取列表页失败: {e}")
            total_fail += 1
            continue

        items = parse_list_page(html)
        if not items:
            print(f"[INFO] 第{page}页无数据，跳过")
            continue
        print(f"  找到 {len(items)} 条")

        for title, date_str, detail_url in items:
            # Check age
            if date_str and date_str < THRESHOLD:
                print(f"  [SKIP] {date_str} {title[:30]} (超3年)")
                total_skip += 1
                continue

            # Fetch detail page
            time.sleep(1)
            try:
                r2 = session.get(detail_url, timeout=20)
                r2.encoding = "utf-8"
            except Exception as e:
                print(f"  [FAIL] 获取详情失败: {detail_url} - {e}")
                total_fail += 1
                continue

            detail_title, detail_date, content, attachments = parse_detail_page(r2.text, detail_url)

            title = detail_title or title
            if detail_date and not date_str:
                date_str = detail_date
            elif detail_date and date_str:
                date_str = detail_date

            if not content.strip():
                # Content might be just attachments
                if attachments:
                    content_parts = []
                    for name, url in attachments:
                        content_parts.append(f"📎 [{name}]({url})")
                    content = "\n\n".join(content_parts)
                else:
                    print(f"  [SKIP] 正文为空: {title[:30]}")
                    total_skip += 1
                    continue

            r = store_item(title, date_str, content, detail_url)
            if r > 0:
                total_new += 1
                print(f"  [OK] +{r} {date_str} {title[:30]} 附件{len(attachments)}")
            else:
                print(f"  [DUP] {date_str} {title[:30]} (已存在)")

    print(f"\n=== 完成 ===")
    print(f"新增: {total_new}, 跳过(超3年): {total_skip}, 失败: {total_fail}")
    return total_new


if __name__ == "__main__":
    main()
