#!/usr/bin/env python3
"""
江苏南大环保科技 - 公示文件 爬虫
URL: http://www.nuep.com.cn/inside/2/87.html
CMS: 自定义ASP.NET CMS
列表: div.n-lb2 > a[href] (title属性=标题, text含日期)
分页: page2.html, page3.html, page4.html
详情: /detail/ID.html
正文: div.n-detail > p (含span内嵌套的p)
"""

import re, sys, time
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "http://www.nuep.com.cn/inside/2/87.html"
SITE_NAME = "江苏南大环保-公示文件"
GROUP = "企业-江苏"
HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
TIMEOUT = 15
DELAY = 1.0
DB_PATH = "/root/search.db"

def parse_date(text):
    m = re.search(r'(\d{4})-(\d{1,2})-(\d{1,2})', text)
    if m:
        return f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return ""

def get_soup(url):
    for retry in range(3):
        try:
            resp = requests.get(url, headers=HEADERS, timeout=TIMEOUT)
            resp.encoding = 'utf-8'
            if resp.status_code == 200:
                return BeautifulSoup(resp.text, 'html.parser')
        except:
            time.sleep(2)
    return None

def fetch_page(page_num):
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"http://www.nuep.com.cn/inside/2/87/page{page_num}.html"
    
    soup = get_soup(url)
    if not soup:
        return []
    
    items = []
    lb = soup.find('div', class_='n-lb2')
    if not lb:
        return items
    
    for a in lb.find_all('a', href=True):
        href = urljoin(BASE_URL, a['href'])
        title = a.get('title', '').strip()
        txt = a.text.strip()
        
        if not title:
            m = re.match(r'(.+?)\s*\d{4}-\d{1,2}-\d{1,2}', txt)
            if m:
                title = m.group(1).strip()
            else:
                title = txt[:60]
        
        date_str = parse_date(txt)
        items.append({'title': title, 'url': href, 'date': date_str})
    
    return items

def get_total_pages(soup):
    max_p = 0
    for a in soup.find_all('a'):
        txt = a.text.strip()
        if txt.isdigit():
            n = int(txt)
            if n > max_p:
                max_p = n
    return max_p or 4

def extract_content(soup):
    title = ""
    pub_date = ""
    parts = []
    attachments = []
    
    nd = soup.find('div', class_='n-detail')
    if not nd:
        t = soup.find('title')
        return (t.text.strip() if t else "", "", "", [])
    
    # Title
    h2 = nd.find('h2')
    if h2:
        title = h2.text.strip()
    if not title:
        t = soup.find('title')
        if t:
            title = t.text.strip()
    
    # Date
    share = nd.find('div', class_='fenxiang')
    if share:
        m = re.search(r'\u53d1\u5e03\u65f6\u95f4\uff1a(\d{4}-\d{1,2}-\d{1,2})', share.text)
        if m:
            pub_date = m.group(1)
    
    # Content - extract all <p> tags recursively from n-detail
    for p in nd.find_all('p'):
        txt = p.get_text(strip=True)
        if txt and not all(c in ' \n\r\t\u3000\xa0' for c in txt):
            parts.append(txt)
    
    # Also check non-p text nodes
    for child in nd.children:
        if not child.name:
            t = str(child).strip()
            if t and len(t) > 5 and not all(c in ' \n\r\t\u3000\xa0' for c in t):
                parts.append(t)
    
    # Attachments
    for a in nd.find_all('a', href=True):
        h = a['href']
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ceb|ofd|ppt|pptx)$', h.lower()):
            attachments.append(f"[{a.text.strip()}]({urljoin(BASE_URL, h)})")
    
    # Global attachment scan
    for a in soup.find_all('a', href=True):
        h = a['href']
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ceb|ofd|ppt|pptx)$', h.lower()):
            entry = f"[{a.text.strip()}]({urljoin(BASE_URL, h)})"
            if entry not in attachments:
                attachments.append(entry)
    
    # Dedup
    seen = set()
    up = []
    for p in parts:
        k = p[:100]
        if k not in seen:
            seen.add(k)
            up.append(p)
    
    return title, '\n\n'.join(up), pub_date, attachments

def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    c.execute('''CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        page_url TEXT, title TEXT, content TEXT, publish_date TEXT,
        site_name TEXT, summary TEXT, attachments TEXT, date_rank TEXT,
        created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP
    )''')
    c.execute('''CREATE INDEX IF NOT EXISTS idx_gov_raw_url ON gov_raw(page_url, site_name)''')
    conn.commit()
    conn.close()

def save_to_db(items):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    n = 0
    for item in items:
        summary = item['content'][:200] if item['content'] else ''
        try:
            c.execute('''INSERT OR REPLACE INTO gov_raw (page_url, title, content, publish_date, site_name, summary, attachments, date_rank, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, \'crawl_nuep.py\')''', (
                item['url'], item['title'], item['content'], item['pub_date'],
                SITE_NAME, summary,
                '\n'.join(item['attachments']) if item['attachments'] else '',
                item['pub_date'] or '0000-00-00',
            ))
            n += 1
        except:
            pass
    conn.commit()
    conn.close()
    return n

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--pages', type=int, default=0)
    parser.add_argument('--skip-db', action='store_true')
    args = parser.parse_args()
    
    init_db()
    print(f"===== {SITE_NAME} =====")
    
    items = fetch_page(1)
    if not items:
        print("No items")
        return
    
    soup = get_soup(BASE_URL)
    total_pages = get_total_pages(soup) if soup else 4
    print(f"Total pages: {total_pages}")
    
    pages = total_pages if args.pages == 0 else min(args.pages, total_pages)
    print(f"Will fetch: {pages}")
    
    all_items = []
    for page in range(1, pages + 1):
        page_items = fetch_page(page)
        if not page_items:
            print(f"\n--- Page {page}/{pages} --- no items")
            continue
        
        print(f"\n--- Page {page}/{pages} ({len(page_items)} items) ---")
        for idx, item in enumerate(page_items):
            print(f"  [{len(all_items)+1}] {item['title'][:40]}...")
            
            ds = get_soup(item['url'])
            if not ds:
                all_items.append(dict(item, content='', pub_date=item['date'], attachments=[]))
                print(f"    FAIL")
                continue
            
            title, content, pub_date, attachments = extract_content(ds)
            all_items.append({
                'title': title or item['title'],
                'url': item['url'],
                'content': content,
                'pub_date': pub_date or item['date'],
                'attachments': attachments,
            })
            
            if content:
                print(f"    seg={len(content.split(chr(10)+chr(10)))} | attach={'YES' if attachments else 'no'}")
            else:
                print(f"    WARN empty")
            
            time.sleep(DELAY)
    
    t = len(all_items)
    wb = sum(1 for i in all_items if i['content'])
    ta = sum(len(i['attachments']) for i in all_items)
    avg = sum(len(i['content'].split('\n\n')) for i in all_items if i['content']) / max(wb, 1)
    
    print(f"\n===== DONE =====")
    print(f"Total: {t}")
    if t:
        print(f"With body: {wb} ({wb/t*100:.1f}%)")
    print(f"Avg seg: {avg:.1f}")
    print(f"Attachments: {ta}")
    
    if not args.skip_db:
        print(f"Imported: {save_to_db(all_items)}")

if __name__ == '__main__':
    main()
