#!/usr/bin/env python3
"""
陕西中汇煤化 - 公司要闻爬虫
http://www.sxzhcc.cn/index.php?c=category&id=14
CMS: FineCMS
"""
import sys, re, json, time, requests, sqlite3
from bs4 import BeautifulSoup
import urllib3
urllib3.disable_warnings()

DB_PATH = "/root/search.db"
SITE_NAME = "陕西中汇煤化-公司要闻"
CATEGORY = "新闻"
GROUP = "企业"
BASE = "http://www.sxzhcc.cn"
LIST_URL = "/index.php?c=category&id=14"
PAGINATION_TPL = "/index.php?c=category&id=14&page={}"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}


def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=30)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=30000")
    return conn


def save_article(conn, title, page_url, publish_date, content):
    summary = content[:200] if content else title
    summary = re.sub(r"\s+", " ", summary).strip()
    try:
        cur = conn.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date, summary, content, category, group_name)
               VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
            (SITE_NAME, page_url, page_url, title.strip(), publish_date,
             summary, content, CATEGORY, GROUP),
        )
        return cur.rowcount > 0
    except Exception as e:
        print(f"  DB error: {e}", file=sys.stderr)
        return False


def crawl_list(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        return [], False
    if resp.status_code != 200:
        return [], False

    soup = BeautifulSoup(resp.text, 'html.parser')
    articles = []
    ul = soup.find('div', class_='plist')
    if not ul:
        return [], False

    for li in ul.find_all('li'):
        a_tag = li.find('a', href=True)
        if not a_tag:
            continue
        href = a_tag.get('href', '')
        title = a_tag.get('title', '') or a_tag.get_text(strip=True)
        date_span = a_tag.find('span')
        date_text = date_span.get_text(strip=True) if date_span else ''
        if href.startswith('/'):
            href = BASE + href
        if not title or len(title) < 5:
            continue
        articles.append({'title': title, 'url': href, 'date': date_text})

    page_div = soup.find('ul', class_='pagination')
    has_next = bool(page_div and page_div.find('a', string=re.compile(r'下一页|最后一页')))
    return articles, has_next


def crawl_detail(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30, verify=False)
        resp.encoding = 'utf-8'
    except Exception as e:
        return None, None, None
    if resp.status_code != 200:
        return None, None, None

    soup = BeautifulSoup(resp.text, 'html.parser')

    # Title from .title or <title>
    title = None
    title_div = soup.find('div', class_='title')
    if title_div:
        t = title_div.get_text(strip=True)
        if t and len(t) > 5:
            title = t
    if not title:
        title_tag = soup.find('title')
        if title_tag:
            parts = title_tag.get_text(strip=True).split('_')
            if parts and parts[0].strip():
                title = parts[0].strip()

    # Date
    date_text = None
    author_div = soup.find('div', class_='author')
    if author_div:
        for span in author_div.find_all('span'):
            if span.find('i', class_=re.compile(r'clock|bi-clock')):
                dt = span.get_text(strip=True)
                m = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', dt)
                if m:
                    date_text = m.group(1).replace('/', '-')

    # Content
    content_div = soup.find('div', class_='ctext')
    if not content_div:
        return title, date_text or '', None

    parts = []
    for elem in content_div.find_all(['p', 'table'], recursive=True):
        if elem.name == 'p':
            if elem.find_parent('table'):
                continue
            text = elem.get_text('', strip=True)
            if text:
                parts.append(text)
        elif elem.name == 'table':
            parts.append(str(elem))

    content = '\n\n'.join(parts) if parts else ''
    if len(content.strip()) < 20:
        content = None

    return title, date_text or '', content


def main():
    max_pages = MAX_PAGES
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
        except ValueError:
            if sys.argv[1] in ('--incremental',):
                max_pages = 1

    print(f"[{SITE_NAME}] max pages: {max_pages}", file=sys.stderr)
    conn = init_db()
    total_new = 0

    # Clean old data
# QC20260925 去掉整站清空再重灌(抢锁+中途死掉会清空整站; page_url 有 UNIQUE 索引，插入本就幂等)     conn.execute("DELETE FROM gov_raw WHERE site_name = ?", (SITE_NAME,))
    conn.commit()
    print(f"  Cleaned old data", file=sys.stderr)

    # Crawl list pages
    all_articles = []
    for page in range(max_pages):
        if page == 0:
            url = BASE + LIST_URL
        else:
            url = BASE + PAGINATION_TPL.format(page + 1)
        print(f"[LIST] Page {page+1}: {url}", file=sys.stderr)
        items, has_next = crawl_list(url)
        print(f"  Found {len(items)} articles", file=sys.stderr)
        if not items:
            break
        all_articles.extend(items)
        if not has_next:
            break
        time.sleep(1)

    print(f"[LIST] Total: {len(all_articles)}", file=sys.stderr)

    for i, a in enumerate(all_articles):
        print(f"[{i+1}/{len(all_articles)}] {a['title'][:40]}...", file=sys.stderr)

        # Check if exists
        cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (a['url'],))
        if cur.fetchone():
            print(f"  skip (exists)", file=sys.stderr)
            continue

        det_title, det_date, det_content = crawl_detail(a['url'])
        final_title = det_title or a['title']
        final_date = det_date or a['date']
        final_content = det_content or f"[{final_title}]({a['url']})"

        if save_article(conn, final_title, a['url'], final_date, final_content):
            total_new += 1
            if total_new <= 5:
                print(f"  + {final_date} {final_title[:50]}", file=sys.stderr)

        time.sleep(0.5)

    conn.commit()
    conn.close()
    print(f"\nDone: {total_new} new articles inserted (total {len(all_articles)} found)", file=sys.stderr)
    print(total_new)


if __name__ == '__main__':
    main()
