#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
孝义市梧桐煤化工园区产业服务平台 - 园区动态 爬虫
CMS: Portal-SaaS (中企动力)
列表: /news/1574549092943745024-{offset}-20.html (20条/页)
详情: /news_detailed/c-_detailId%3D{id}.html
"""

import requests
import re
import json
import time
import sys
import os
import sqlite3
from bs4 import BeautifulSoup

BASE_URL = "http://www.xymhgfw.com"
SITE_NAME = "孝义市梧桐煤化工园区-园区动态"
GROUP = "企业"
DB_PATH = os.environ.get('DB_PATH', '/root/search.db')
MAX_PAGES = 5  # 前5页
CATEGORY_ID = "1574549092943745024"
PAGE_SIZE = 20

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

session = requests.Session()
session.headers.update(HEADERS)


def fetch(url, timeout=30):
    resp = session.get(url, timeout=timeout)
    resp.encoding = 'utf-8'
    return resp


def crawl_list(page_num):
    """爬取列表页"""
    offset = (page_num - 1) * PAGE_SIZE
    url = f"{BASE_URL}/news/{CATEGORY_ID}-{offset}-{PAGE_SIZE}.html"
    
    print(f"  Fetching page {page_num}: {url}")
    
    try:
        resp = fetch(url)
        resp.raise_for_status()
    except Exception as e:
        print(f"  Error: {e}")
        return []
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    items = []
    
    # 列表项分析：portal-saas 列表结构
    # 方法1: 找所有链接到 news_detailed 的 a 标签
    for a_tag in soup.find_all('a', href=re.compile(r'/news_detailed/')):
        href = a_tag.get('href', '')
        if not href or 'javascript' in href:
            continue
        
        title = a_tag.get_text(strip=True)
        if not title:
            continue
        
        if not href.startswith('http'):
            href = f"{BASE_URL}{href}" if href.startswith('/') else f"{BASE_URL}/{href}"
        
        # 日期 - 在附近找
        parent = a_tag.find_parent(['li', 'div'])
        date = ''
        if parent:
            date_match = re.search(r'(\d{4}[-/]\d{1,2}[-/]\d{1,2})', parent.get_text())
            if date_match:
                date = date_match.group(1)
        
        items.append({
            'title': title,
            'url': href,
            'date': date,
        })
    
    return items


def crawl_detail(item):
    """爬取详情页"""
    url = item['url']
    print(f"    Detail: {item['title'][:40]}...")
    
    try:
        resp = fetch(url)
        resp.raise_for_status()
    except Exception as e:
        print(f"    Error: {e}")
        return None
    
    soup = BeautifulSoup(resp.text, 'html.parser')
    
    # 标题 - 从 <title> 标签（格式: "文章标题-站点名称"）
    title = item['title']
    title_tag = soup.find('title')
    if title_tag:
        t_text = title_tag.get_text(strip=True)
        # 去掉站点后缀
        t_text = re.sub(r'-孝义市梧桐煤化工园区产业服务平台$', '', t_text).strip()
        if t_text:
            title = t_text
    
    # 如果 <title> 没有，从第2个 <h1> 取（第1个是站点名）
    if not title:
        h1_tags = soup.find_all('h1')
        if len(h1_tags) > 1:
            h1_text = h1_tags[1].get_text(strip=True)
            if h1_text:
                title = h1_text
    
    # 日期
    date = item['date']
    date_tag = soup.find('p', class_=re.compile(r'timeFormat|time'))
    if date_tag:
        d = date_tag.get_text(strip=True)
        date_match = re.search(r'(\d{4}-\d{2}-\d{2})', d)
        if date_match:
            date = date_match.group(1)
    if not date:
        # Try span/em
        date_el = soup.find(class_=re.compile(r'time|date'))
        if date_el:
            d = date_el.get_text(strip=True)
            date_match = re.search(r'(\d{4}-\d{2}-\d{2})', d)
            if date_match:
                date = date_match.group(1)
    
    # 正文
    content_parts = []
    attachments = []
    
    # 正文区域: reset_style（唯一，包含文章正文），或 e_richText-7
    content_div = soup.find('div', class_='reset_style')
    if not content_div:
        content_div = soup.find('div', class_=re.compile(r'reset_style'))
    if not content_div:
        content_div = soup.find('div', class_=re.compile(r'e_richText-7'))

    if content_div:
        # 附件链接
        for a_tag in content_div.find_all('a', href=True):
            href = a_tag['href']
            if re.search(r'\.(doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar)$', href, re.I):
                attach_title = a_tag.get_text(strip=True)
                if not href.startswith('http'):
                    href = f"{BASE_URL}{href}" if href.startswith('/') else f"{url.rstrip('/')}/{href}"
                attachments.append({
                    'title': attach_title or os.path.basename(href),
                    'url': href
                })
        
        # 正文段落
        for p in content_div.find_all(['p', 'div']):
            if not p.get_text(strip=True):
                continue
            if p.find_parent(['form', 'textarea', 'script']):
                continue
            
            text = p.get_text(strip=True)
            if re.match(r'^(责任编辑|初审|复审|终审|编辑[：:]|来源[：:])', text):
                continue
            
            # 保留HTML表格
            tables = p.find_all('table')
            if tables:
                for table in tables:
                    content_parts.append(str(table))
                continue
            
            content_parts.append(text)
    
    content = '\n\n'.join(part for part in content_parts if part)
    
    # 空内容回退
    if len(content.strip()) < 20:
        attach_links = '\n'.join(f"[{a['title']}]({a['url']})" for a in attachments)
        if attach_links:
            content = f'<p><a href="{url}">{title}</a></p>\n\n附件：\n{attach_links}'
        else:
            content = f'<p><a href="{url}">{title}</a></p>'
    
    summary = content[:200] if content else title
    
    return {
        'site_name': SITE_NAME,
        'source_url': '',
        'page_url': url,
        'title': title,
        'publish_date': date,
        'content': content,
        'summary': summary,
        'category': GROUP,
        'status': 'published',
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '',
    }


def save_to_db(records, db_path=DB_PATH):
    """保存到数据库"""
    if not records:
        return 0
    
    conn = sqlite3.connect(db_path, timeout=60)
    c = conn.cursor()
    
    new_count = 0
    for i, rec in enumerate(records):
        try:
            c.execute('''INSERT OR IGNORE INTO gov_raw 
                (site_name, source_url, page_url, title, publish_date,
                 content, summary, category, status, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)''',
                (rec['site_name'], rec['source_url'], rec['page_url'],
                 rec['title'], rec['publish_date'], rec['content'],
                 rec['summary'], rec['category'], rec['status'],
                 rec['attachments']))
            if c.rowcount > 0:
                new_count += 1
                if new_count <= 5 or (i+1) % 20 == 0:
                    print(f"  [{i+1}/{len(records)}] {rec['title'][:40]} +{new_count}")
        except Exception as e:
            print(f"  ERROR [{rec['page_url']}]: {e}")
    
    conn.commit()
    conn.close()
    return new_count


def main():
    print(f"=== {SITE_NAME} - 全量爬取 ===")
    
    all_items = []
    seen_urls = set()
    
    for page in range(1, MAX_PAGES + 1):
        items = crawl_list(page)
        if not items:
            print(f"  -> 0 items on page {page}, stopping")
            break
        print(f"  -> {len(items)} items on page {page}")
        for item in items:
            if item['url'] not in seen_urls:
                seen_urls.add(item['url'])
                all_items.append(item)
    
    print(f"\nTotal unique items: {len(all_items)}")
    
    records = []
    for idx, item in enumerate(all_items):
        print(f"  [{idx+1}/{len(all_items)}] {item['title'][:30]}...")
        detail = crawl_detail(item)
        if detail:
            records.append(detail)
        time.sleep(0.3)
    
    if records:
        count = save_to_db(records)
        print(f"\nInserted {count} new records (trigger auto-syncs FTS)")
    else:
        print("\nNo records to insert")


if __name__ == '__main__':
    if '--incremental' in sys.argv:
        MAX_PAGES = 1
    main()
