#!/usr/bin/env python3
"""
crawl_hexgroup.py — 河北惠尔信新材料股份有限公司-资料公示
Site: https://www.hexgroup.cn/news/4/
CMS: 中企动力 SaaS (portal-saas)
List: div.cbox-2.p_loopitem > p.e_text-5 > a
Detail: div.e_richText-23.s_title.clearfix
Pagination: /news_list/1777254464689545216-{from}-10.html
"""
import re, os, sys, time, sqlite3, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
SITE_NAME = "hexgroup.cn-资料公示"
BASE_URL = "https://www.hexgroup.cn"
LIST_URL = BASE_URL + "/news/4/"
PAGE_URL_TPL = BASE_URL + "/news_list/1777254464689545216-{}-10.html"
CUTOFF = (datetime.now() - timedelta(days=3*365)).strftime("%Y-%m-%d")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Referer": BASE_URL + "/",
}

session = requests.Session()
session.headers.update(HEADERS)


def fetch(url):
    try:
        r = session.get(url, timeout=30)
        r.encoding = 'utf-8'
        return r.text
    except Exception as e:
        print(f"  [ERR] fetch failed: {url} - {e}")
        return None


def parse_list(html):
    """Parse list page, return list of (url, title, date_str)"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    for item_div in soup.select('div.cbox-2.p_loopitem'):
        # Title link
        a = item_div.select_one('p.e_text-5.s_title a[href]')
        if not a:
            continue
        href = a.get('href', '')
        title = a.get_text(strip=True)
        if not href or not title:
            continue
        if not href.startswith('http'):
            href = BASE_URL + href

        # Date from p.e_timeFormat-9.s_title
        date_el = item_div.select_one('p.e_timeFormat-9.s_title')
        date_str = ''
        if date_el:
            date_str = date_el.get_text(strip=True)
            # Format: 2026/01/14 -> 2026-01-14
            date_str = date_str.replace('/', '-')

        items.append((href, title, date_str))
    return items


def parse_detail(html, url):
    """Parse detail page, return (title, publish_date, content)"""
    soup = BeautifulSoup(html, 'html.parser')

    # Title from <title>, strip site suffix
    title_tag = soup.find('title')
    title = ''
    if title_tag:
        raw = title_tag.get_text(strip=True)
        # Strip "-河北惠尔信新材料股份有限公司" suffix
        title = re.sub(r'\s*[-–—]\s*河北惠尔信新材料股份有限公司\s*$', '', raw).strip()

    # Content from div.e_richText-23.s_title.clearfix
    content_div = soup.select_one('div.e_richText-23.s_title.clearfix')
    content = ''
    if content_div:
        content = str(content_div)
        content = content.strip()

    # Date: try detail page first
    date_str = ''
    # Look for YYYY-MM-DD followed by </P> (the <P> may contain SVG before the date text)
    m = re.search(r'(\d{4}-\d{2}-\d{2})\s*</P>', html)
    if m:
        date_str = m.group(1)
    # Also try </p> as fallback
    if not date_str:
        m = re.search(r'(\d{4}-\d{2}-\d{2})\s*</p>', html)
        if m:
            date_str = m.group(1)

    return title, date_str, content


def get_total_pages():
    """Get total pages by checking page 1 links"""
    html = fetch(LIST_URL)
    if not html:
        return 1

    soup = BeautifulSoup(html, 'html.parser')
    from_values = set()
    for a in soup.find_all('a', href=True):
        href = a['href']
        m = re.search(r'/news_list/\d+-(\d+)-\d+\.html', href)
        if m:
            from_values.add(int(m.group(1)))

    if not from_values:
        return 1

    max_from = max(from_values)
    return (max_from // 10) + 1  # size=10


def main():
    print(f"=== {SITE_NAME} ===")
    print(f"Cutoff: {CUTOFF}")

    total_pages = get_total_pages()
    print(f"Total pages: {total_pages}")

    conn = sqlite3.connect(SEARCH_DB, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    cur = conn.cursor()

    total_new = 0
    total_skip = 0

    for page in range(1, total_pages + 1):
        from_val = (page - 1) * 10
        list_url = PAGE_URL_TPL.format(from_val) if page > 1 else LIST_URL
        print(f"\nPage {page}/{total_pages}: {list_url}")

        html = fetch(list_url)
        if not html:
            print(f"  [SKIP] page {page} fetch failed")
            continue

        items = parse_list(html)
        if not items:
            print(f"  [SKIP] page {page} no items")
            continue

        print(f"  Found {len(items)} items")

        page_new = 0
        for url, title, date_str in items:
            # Check if already in DB
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                total_skip += 1
                continue

            # Filter by date from list page
            if date_str and date_str < CUTOFF:
                total_skip += 1
                continue

            # Fetch detail
            detail_html = fetch(url)
            if not detail_html:
                total_skip += 1
                continue

            full_title, detail_date, content = parse_detail(detail_html, url)

            if not full_title:
                full_title = title

            # Date: prefer detail page date, fallback to list page
            final_date = detail_date or date_str
            if not final_date:
                total_skip += 1
                continue
            if final_date < CUTOFF:
                total_skip += 1
                continue

            # Content validation
            content = content.strip()
            if not content:
                total_skip += 1
                continue

            # Summary
            summary = ''
            if content:
                text_soup = BeautifulSoup(content, 'html.parser')
                plain = text_soup.get_text(strip=True)
                summary = plain[:200]

            try:
                cur.execute(
                    "INSERT OR IGNORE INTO gov_raw "
                    "(site_name, title, page_url, publish_date, content, date_rank, category) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?)",
                    (SITE_NAME, full_title, url, final_date, content,
                     int(final_date.replace("-", "")), 'tzgg')
                )
                if cur.rowcount > 0:
                    page_new += 1
                    total_new += 1
            except Exception as e:
                print(f"  [ERR] insert: {e}")

        conn.commit()
        print(f"  Page {page}: +{page_new} new (total {total_new}, skip {total_skip})")

        time.sleep(0.5)

    conn.close()
    print(f"\n=== Done: {total_new} new, {total_skip} skipped ===")


if __name__ == "__main__":
    main()
