#!/usr/bin/env python3
"""忠县-生态环境局-建设项目环评审批情况 爬虫 (WCM CMS) - 修复表格"""
import requests, re, sqlite3, sys, os, time
from bs4 import BeautifulSoup, Tag
from urllib.parse import urljoin

VERIFY = False
requests.packages.urllib3.disable_warnings()
BASE = 'http://www.zhongxian.gov.cn/bm/zxsthjj/zwgk_34581/fdzdgknr_34586/zdxmhjyxpj/jsxmhpspqk'
HEADERS = {'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36'}
DB = '/root/search.db'
SITE_NAME = '忠县_环评审批情况'

def get_text(soup):
    return soup.get_text(separator=' ', strip=True) if soup else ''

def extract_content(html):
    soup = BeautifulSoup(html, 'html.parser')
    article = soup.find('div', class_='detail-article')
    if not article:
        return '', []
    
    # Attachments
    attachments = []
    for a in article.find_all('a', href=True):
        h = a['href']
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', h, re.I):
            attachments.append(urljoin(BASE, h))
    
    parts = []
    
    # 1. Extract paragraphs NOT inside <table>
    non_table_paragraphs = []
    for p in article.find_all('p'):
        # Check if this <p> is inside a <table>
        parent = p.parent
        is_in_table = False
        while parent:
            if parent.name == 'table':
                is_in_table = True
                break
            parent = parent.parent if hasattr(parent, 'parent') else None
        
        if not is_in_table:
            txt = p.get_text(strip=True)
            if txt:
                non_table_paragraphs.append(txt)
    
    if non_table_paragraphs:
        parts.append('\n\n'.join(non_table_paragraphs))
    
    # 2. Extract <table> as HTML (preserve structure)
    for table in article.find_all('table'):
        html_table = str(table)
        if html_table and len(html_table) > 50:
            parts.append(html_table)
    return '\n\n'.join(parts), attachments


def parse_list_page(url):
    try:
        r = requests.get(url, headers=HEADERS, verify=VERIFY, timeout=30)
        r.encoding = 'utf-8'
        if len(r.text) < 500:
            return [], 0
    except Exception as e:
        print(f'[ERROR] {url}: {e}', flush=True)
        return [], 0
    
    soup = BeautifulSoup(r.text, 'html.parser')
    items = []
    
    total = 0
    m = re.search(r'wcm分页，总记录：(\d+)', r.text)
    if m:
        total = int(m.group(1))
    
    news_list = soup.find('ul', class_='news-list')
    if not news_list:
        return [], total
    
    for li in news_list.find_all('li'):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        href = urljoin(url, a['href'])
        title = get_text(a)
        if not title:
            continue
        span = li.find('span')
        date = get_text(span) if span else ''
        items.append((title, href, date))
    
    return items, total


def get_existing_urls():
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    urls = set()
    for row in c.execute("SELECT page_url FROM gov_raw WHERE site_name=?", (SITE_NAME,)):
        urls.add(row[0])
    conn.close()
    return urls


def save_batch(items):
    if not items:
        return 0
    conn = sqlite3.connect(DB, timeout=60)
    c = conn.cursor()
    saved = 0
    for item in items:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, content, summary, publish_date, site_name) VALUES (?,?,?,?,?,?)",
                (item['title'], item['url'], item['content'], item['content'][:500], item['pub_date'], SITE_NAME)
            )
            if c.rowcount > 0:
                conn.commit()
                saved += 1
        except Exception as e:
            print(f'[DB ERROR] {e}', flush=True)
            conn.rollback()
    conn.close()
    return saved


def get_page_url(page):
    if page == 1:
        return BASE + '/index.html'
    else:
        return BASE + '/index_{}.html'.format(page)


def main():
    existing_urls = get_existing_urls()
    print('[START] Already in DB: {}'.format(len(existing_urls)), flush=True)
    
    items, total = parse_list_page(get_page_url(1))
    if not items and total == 0:
        print('[ERROR] Cannot parse list page', flush=True)
        return
    
    items_per_page = len(items) if items else 15
    total_pages = max(1, (total + items_per_page - 1) // items_per_page) if total else 30
    print('[INFO] Total records: {}, items/page: {}, total pages: ~{}'.format(total, items_per_page, total_pages), flush=True)
    
    all_list_items = list(items)
    batch = []
    total_new = 0
    
    for page in range(2, total_pages + 1):
        url = get_page_url(page)
        print('[PAGE {}/{}]'.format(page, total_pages), flush=True)
        page_items, _ = parse_list_page(url)
        if not page_items:
            print('  Empty page {}, stopping'.format(page), flush=True)
            break
        all_list_items.extend(page_items)
        print('  {} items'.format(len(page_items)), flush=True)
        time.sleep(0.5)
    
    print('[TOTAL] {} items from list'.format(len(all_list_items)), flush=True)
    
    for i, (title, detail_url, date) in enumerate(all_list_items, 1):
        if detail_url in existing_urls:
            continue
        
        try:
            r = requests.get(detail_url, headers=HEADERS, verify=VERIFY, timeout=30)
            r.encoding = 'utf-8'
        except Exception as e:
            print('  [SKIP] {} - {}'.format(title[:40], e), flush=True)
            continue
        
        content, attachments = extract_content(r.text)
        if not content:
            print('  [SKIP] No content: {}'.format(title[:40]), flush=True)
            continue
        
        batch.append({
            'title': title,
            'url': detail_url,
            'pub_date': date,
            'content': content,
            'attachments': '\n'.join(attachments) if attachments else ''
        })
        
        if len(batch) >= 20:
            saved = save_batch(batch)
            total_new += saved
            print('  [{}/{}] Saved {}, cumulative: {}'.format(i, len(all_list_items), saved, total_new), flush=True)
            batch = []
        
        time.sleep(0.3)
    
    if batch:
        saved = save_batch(batch)
        total_new += saved
    
    print('[DONE] New: {}, Total: {}'.format(total_new, len(existing_urls) + total_new), flush=True)


if __name__ == '__main__':
    main()
