#!/usr/bin/env python3
"""
宁夏石油化工环境科学院股份有限公司 - 公示公告 爬虫
URL: http://www.nxshhky.com/news/class/?86.html
CMS: 自定义PHP CMS (PDV)
列表: div.title > a + div (date sibling)
分页: index.php?page=N&catid=86&myord=dtime&myshownums=9
详情: /news/html/?ID.html
正文: div.con
"""

import re, sys, time, os
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime

BASE_URL = "http://www.nxshhky.com/news/class/?86.html"
SITE_NAME = "宁夏石油化工环境科学院-公示公告"
GROUP = "企业-宁夏"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
}
TIMEOUT = 15
DELAY = 1.0

DB_PATH = "/root/search.db"

def parse_date(text):
    m = re.search(r'(\d{4})-(\d{1,2})-(\d{1,2})', text)
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return ""

def get_soup(url):
    for retry in range(3):
        try:
            resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return BeautifulSoup(resp.text, 'html.parser')
        except Exception as e:
            pass
            time.sleep(2)
    return None

def fetch_list_page(page_num):
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"http://www.nxshhky.com/news/class/index.php?page={page_num}&catid=86&myord=dtime&myshownums=9&showtj=&showdate=&author=&key="
    soup = get_soup(url)
    if not soup:
        return []
    
    items = []
    for a in soup.find_all('a', href=True):
        href = a['href']
        if '../news/html/' in href or '/news/html/?' in href:
            link = urljoin(BASE_URL, href)
            title = a.text.strip()
            if not title:
                continue
            
            # Date from next sibling div
            p = a.parent
            date_str = ""
            for s in p.find_next_siblings():
                if s.name == 'div':
                    txt = s.text.strip()
                    d = parse_date(txt)
                    if d:
                        date_str = d
                        break
                elif s.name and s.get_text(strip=True):
                    txt = s.get_text(strip=True)
                    d = parse_date(txt)
                    if d:
                        date_str = d
                        break
            
            items.append({
                'title': title,
                'url': link,
                'date': date_str,
            })
    
    return items

def get_total_pages(soup):
    max_page = 0
    for a in soup.find_all('a'):
        txt = a.text.strip()
        if txt.isdigit():
            n = int(txt)
            if n > max_page:
                max_page = n
    return max_page or 10

def extract_content(soup):
    title = ""
    t = soup.find('title')
    if t:
        title = t.text.strip()
        title = re.sub(r'-\u5b81\u590f\u77f3\u6cb9\u5316\u5de5\u73af\u5883\u79d1\u5b66\u9662\u80a1\u4efd\u6709\u9650\u516c\u53f8$', '', title).strip()
    
    pub_date = ""
    pdv = soup.find('div', class_='pdv_content')
    if pdv:
        txt = pdv.text.strip()
        m = re.search(r'(\d{4}-\d{1,2}-\d{1,2})\s+\d{1,2}:\d{2}:\d{2}', txt)
        if m:
            pub_date = m.group(1)
    
    parts = []
    attachments = []
    
    con = soup.find('div', class_='con')
    if not con:
        con = soup.find('div', id='con')
    
    if con:
        for child in con.children:
            if not child.name:
                text = str(child).strip()
                if text and not all(c in ' \n\r\t\u3000\xa0' for c in text):
                    parts.append(text)
                continue
            
            tag = child.name.lower()
            
            if tag in ('p', 'div'):
                has_block = any(c.name in ('table', 'div', 'ul', 'ol') for c in child.find_all(recursive=False))
                if has_block:
                    for sub in child.children:
                        if not sub.name:
                            t = str(sub).strip()
                            if t and not all(c in ' \n\r\t\u3000\xa0' for c in t):
                                parts.append(t)
                            continue
                        if sub.name == 'table':
                            parts.append(str(sub))
                        elif sub.name in ('p', 'div'):
                            sub_txt = sub.get_text(strip=True)
                            if sub_txt:
                                parts.append(sub_txt)
                        else:
                            t = sub.get_text(strip=True)
                            if t:
                                parts.append(t)
                else:
                    txt = child.get_text(strip=True)
                    if txt and not all(c in ' \n\r\t\u3000\xa0' for c in txt):
                        parts.append(txt)
            
            elif tag == 'table':
                parts.append(str(child))
            elif tag == 'img':
                src = child.get('src', '')
                alt = child.get('alt', '')
                if src:
                    parts.append(f"![{alt}]({urljoin(BASE_URL, src)})")
            elif tag in ('ul', 'ol'):
                txt = child.get_text(strip=True)
                if txt:
                    parts.append(txt)
            elif tag == 'br':
                pass
            else:
                txt = child.get_text(strip=True)
                if txt:
                    parts.append(txt)
        
        # Attachments
        for a in con.find_all('a', href=True):
            href = a['href']
            text = a.text.strip()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ceb|ofd|ppt|pptx)$', href.lower()):
                full_url = urljoin(BASE_URL, href)
                attachments.append(f"[{text}]({full_url})")
    
    seen = set()
    unique_parts = []
    for p in parts:
        key = p[:100]
        if key not in seen:
            seen.add(key)
            unique_parts.append(p)
    
    content = '\n\n'.join(unique_parts)
    return title, content, pub_date, attachments

def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        page_url TEXT,
        title TEXT,
        content TEXT,
        publish_date TEXT,
        site_name TEXT,
        summary TEXT,
        attachments TEXT,
        date_rank TEXT,
        created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    c.execute('''CREATE INDEX IF NOT EXISTS idx_gov_raw_url 
        ON gov_raw(page_url, site_name)''')
    conn.commit()
    conn.close()

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    imported = 0
    for item in items:
        summary = item['content'][:200] if item['content'] else ''
        try:
            c.execute('''INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, \'crawl_nxshhky.py\')''', (
                item['url'],
                item['title'],
                item['content'],
                item['pub_date'],
                SITE_NAME,
                summary,
                '\n'.join(item['attachments']) if item['attachments'] else '',
                item['pub_date'] or '0000-00-00',
            ))
            imported += 1
        except Exception as e:
            pass
    conn.commit()
    conn.close()
    return imported

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--pages', type=int, default=0)
    parser.add_argument('--skip-db', action='store_true')
    args = parser.parse_args()
    
    init_db()
    
    print(f"===== {SITE_NAME} 爬取 =====")
    soup = get_soup(BASE_URL)
    if not soup:
        print("FAIL")
        return
    
    total_pages = get_total_pages(soup)
    print(f"Pages: {total_pages}")
    
    pages_to_fetch = total_pages if args.pages == 0 else min(args.pages, total_pages)
    print(f"Will fetch: {pages_to_fetch} pages")
    
    all_items = []
    page_counts = []
    
    for page in range(1, pages_to_fetch + 1):
        print(f"\n--- Page {page}/{pages_to_fetch} ---")
        items = fetch_list_page(page)
        if not items:
            print("  No items")
            continue
        
        page_counts.append(len(items))
        print(f"  Items: {len(items)}")
        
        for idx, item in enumerate(items):
            print(f"  [{len(all_items)+1}] {item['title'][:40]}...")
            
            detail_soup = get_soup(item['url'])
            if not detail_soup:
                print(f"    FAIL detail")
                all_items.append({
                    'title': item['title'],
                    'url': item['url'],
                    'content': '',
                    'pub_date': item['date'],
                    'attachments': [],
                })
                continue
            
            title, content, pub_date, attachments = extract_content(detail_soup)
            final_title = title or item['title']
            final_date = pub_date or item['date']
            
            all_items.append({
                'title': final_title,
                'url': item['url'],
                'content': content,
                'pub_date': final_date,
                'attachments': attachments,
            })
            
            if content:
                seg = len(content.split('\n\n'))
                print(f"    seg={seg} | attach={'YES' if attachments else 'no'}")
            else:
                print(f"    WARN empty")
            
            time.sleep(DELAY)
    
    total = len(all_items)
    wb = sum(1 for i in all_items if i['content'])
    ta = sum(len(i['attachments']) for i in all_items)
    avg = sum(len(i['content'].split('\n\n')) for i in all_items if i['content']) / max(wb, 1)
    
    print(f"\n===== DONE =====")
    print(f"Total: {total}")
    print(f"With body: {wb} ({wb/total*100:.1f}%)" if total else f"With body: 0")
    print(f"Avg seg: {avg:.1f}")
    print(f"Attachments: {ta}")
    if page_counts:
        print(f"Per page avg: {sum(page_counts)/len(page_counts):.1f}")
    
    if not args.skip_db:
        imp = save_to_db(all_items)
        print(f"Imported: {imp}")

if __name__ == '__main__':
    main()
