#!/usr/bin/env python3
"""
兴宁市生态环境局 - 环保信息爬虫
List: https://www.xingning.gov.cn/zfjg/xnshjbhj/hbxx/index.html (+ index_2.html ... index_20.html)
Detail: /zfjg/xnshjbhj/hbxx/*/content/post_XXXXX.html
"""
import requests, re, json, sqlite3, time, os, sys
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = '/root/search.db'
SCRIPT_NAME = 'crawl_xingning.py'
SITE_NAME = '兴宁市-环保信息'
PROVINCE = '广东'
BASE_URL = 'https://www.xingning.gov.cn'

HEADERS = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36',
}

conn = sqlite3.connect(DB_PATH, timeout=60)
cur = conn.cursor()
total_new = 0

def save_article(title, page_url, publish_date, content_text, attachments):
    global total_new
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    try:
        cur.execute("""
            INSERT OR IGNORE INTO gov_raw (title, page_url, publish_date, content, site_name, category, attachments)
            VALUES (?, ?, ?, ?, ?, ?, ?)
        """, (title.strip(), page_url, publish_date, content_text.strip(), SITE_NAME, '环保信息', attachments_json))
        if cur.rowcount > 0:
            total_new += 1
            return True
    except Exception as e:
        pass
    return False

def fetch_detail(url):
    """Fetch detail page and extract content + attachments"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except:
        return None, []
    
    soup = BeautifulSoup(html, 'html.parser')
    
    # Try content div (NFCMS: is-contentbox)
    content_div = soup.find('div', class_='is-contentbox')
    if not content_div:
        content_div = soup.find('div', id=lambda i: i and 'content' in i.lower() if i else None)
    if not content_div:
        content_div = soup.find('div', class_=lambda c: c and 'content' in c.lower() if c else None)
    
    # Fallback: old regex method
    if not content_div:
        return _fetch_detail_old(url, html)
    
    # Remove scripts
    for s in content_div.find_all('script'):
        s.decompose()
    
    # Process tables
    table_tags = content_div.find_all('table')
    table_md_blocks = []
    
    for t in table_tags:
        all_text = t.get_text(strip=True)
        # Skip metadata tables
        if any(kw in all_text for kw in ['浏览次数', '信息来源', '发布时间']):
            t.decompose()
            continue
        # Skip layout tables
        if '\u00d7' in all_text or len(all_text) < 20:
            t.decompose()
            continue
        # 保留 HTML 表格
        rows = t.find_all('tr')
        valid_rows = 0
        for row in rows:
            cells = row.find_all(['td', 'th'])
            if any(c.get_text(strip=True).strip() for c in cells):
                valid_rows += 1
        if valid_rows > 1:
            table_md_blocks.append(str(t))
        t.decompose()
    
    # Get text from content div (tables already removed)
    raw_text = content_div.get_text()
    lines = [l.strip() for l in raw_text.split('\n') if l.strip()]
    
    # Filter noise
    skip_kw = ['【打印本页】', '【关闭窗口】', '浏览次数：', '信息来源：', '转自本网', '来源：']
    filtered = []
    for l in lines:
        if any(s in l for s in skip_kw) or l.startswith('上一条：') or l.startswith('下一条：'):
            continue
        filtered.append(l)
    
    # Build result
    title_tag = content_div.find('div', class_='is-newstitle')
    result = []
    title_text = None
    if title_tag:
        title_text = title_tag.get_text(strip=True)
        result.append(title_text)
    
    for line in filtered:
        if title_text and line == title_text:
            continue
        result.append(line)
    
    # Insert table before "公告期限" or append at end
    md_list = [m for m in table_md_blocks if m]
    if md_list:
        insert_idx = -1
        for i, p in enumerate(result):
            if '公告期限' in p or '联系电话' in p:
                insert_idx = i
                break
        if insert_idx >= 0:
            result.insert(insert_idx, md_list[0])
        else:
            result.append(md_list[0])
    
    content_text = '\n\n'.join(result)
    
    # Attachments
    attachments = []
    for a in content_div.find_all('a', href=True):
        href = a['href']
        name = a.get_text(strip=True) or os.path.basename(href).split('?')[0]
        if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            if href.startswith('/'):
                href = BASE_URL + href
            attachments.append({'name': name, 'url': href})
    
    # Deduplicate
    seen = set()
    unique_attachments = []
    for a in attachments:
        if a['url'] not in seen:
            seen.add(a['url'])
            unique_attachments.append(a)
    
    return content_text, unique_attachments


def _fetch_detail_old(url, html):
    """Fallback: old regex-based extraction"""
    content_text = ''
    clean = re.sub(r'<script[^>]*>[\s\S]*?</script>', '', html)
    clean = re.sub(r'<style[^>]*>[\s\S]*?</style>', '', clean)
    
    def table_to_markdown(m):
        table_html = m.group(0)
        rows = re.findall(r'<tr[^>]*>(.*?)</tr>', table_html, re.DOTALL)
        all_text = re.sub(r'<[^>]+>', '', table_html)
        all_text = re.sub(r'&nbsp;|\s+', ' ', all_text).strip()
        if len(all_text) < 20 or '×' in all_text:
            return '\n'
        # 保留 HTML 表格（仍跳过元数据行）
        keep_rows = []
        for ri, row in enumerate(rows):
            cells = re.findall(r'<(?:td|th)[^>]*>(.*?)</(?:td|th)>', row, re.DOTALL)
            md_cells = []
            for cell in cells:
                cell_text = re.sub(r'<[^>]+>', '', cell)
                cell_text = cell_text.replace('&nbsp;', ' ').strip()
                cell_text = re.sub(r'\s+', ' ', cell_text)
                md_cells.append(cell_text)
            if any(kw in ' '.join(md_cells) for kw in ['浏览次数', '信息来源', '转自']):
                continue
            if not any(c.strip() for c in md_cells):
                continue
            keep_rows.append(row)
        if len(keep_rows) > 1:
            keep_html = '<table>' + ''.join(keep_rows) + '</table>'
            return '\n' + keep_html + '\n'
        return '\n'
    
    clean = re.sub(r'<table[^>]*>.*?</table>', table_to_markdown, clean, flags=re.DOTALL)
    
    body = re.search(r'<body[^>]*>([\s\S]*)</body>', clean)
    if body:
        body_html = body.group(1)
        for cls in ['is-header', 'is-footer', 'footer', 'header', 'nav', 'topbar', 'm-top', 'm-logo', 'is-search', 'foot', 'copyright']:
            body_html = re.sub(r'class="[^"]*' + cls + '[^"]*"[\s\S]*?(?=\n\s*<|$)', '', body_html)
        text = re.sub(r'<[^>]+>', '\n', body_html)
        text = re.sub(r'&nbsp;', ' ', text)
        text = re.sub(r'[ \t]+', ' ', text)
        lines = [l.strip() for l in text.split('\n') if len(l.strip()) > 3]
        skip_words = ['设为首页', '加入收藏', '网站标识码', '粤ICP备', '粤公网安备', '主办：', '维护：',
                     '联系我们', '网站地图', '隐私声明', '关于我们', '网站声明', 'RSS订阅', 'English',
                     '兴宁市人民政府', '政务服务', '部门动态', '当前位置']
        content_lines = []
        for line in lines:
            if line in skip_words or any(s == line for s in skip_words):
                continue
            if any(s in line for s in ['建设项目环境影响评价']):
                continue
            if len(line) > 5:
                content_lines.append(line)
        if content_lines:
            # Fix markdown table blank lines
            content_text = '\n\n'.join(content_lines[:60])
            content_text = re.sub(r'(\| [^\n]+ \|)\n\n(\|)', r'\1\n\2', content_text)
    
    attachments = []
    for m in re.finditer(r'<a[^>]*href="(https?://[^"]+\.(?:pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>([^<]*)</a>', html, re.I):
        url_a = m.group(1)
        name = m.group(2).strip() or os.path.basename(url_a).split('?')[0]
        attachments.append({'name': name, 'url': url_a})
    
    return content_text, attachments

def extract_list_page(html):
    """Extract articles from list page"""
    articles = []
    for m in re.finditer(r'<a[^>]*href="([^"]+)"[^>]*title="([^"]*)"', html):
        href = m.group(1)
        title = m.group(2).strip()
        if 'post_' not in href or len(title) < 10:
            continue
        # Full URL
        if href.startswith('/'):
            href = BASE_URL + href
        elif not href.startswith('http'):
            continue
        
        # Find date
        area = html[m.start():m.start()+300]
        dm = re.search(r'(\d{4}-\d{2}-\d{2})', area)
        date = dm.group(1) if dm else ''
        
        articles.append((title, href, date))
    return articles

# Main: scrape list pages
print("Scraping 兴宁市环保信息...")
new_items = 0
max_pages = 20

for page in range(1, max_pages + 1):
    if page == 1:
        url = f'{BASE_URL}/zfjg/xnshjbhj/hbxx/index.html'
    else:
        url = f'{BASE_URL}/zfjg/xnshjbhj/hbxx/index_{page}.html'
    
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        print(f"  Page {page} error: {e}")
        break
    
    articles = extract_list_page(html)
    if not articles:
        print(f"  Page {page}: no articles, stopping")
        break
    
    for title, page_url, date in articles:
        if not date:
            continue
        # Only take articles from last 3 years
        if date < '2023-01-01':
            continue
        
        # Fetch detail
        content_text, attachments = fetch_detail(page_url)
        if not content_text:
            content_text = f'<p><a href="{page_url}">{title}</a></p>'
        
        saved = save_article(title, page_url, date, content_text, attachments)
    
    print(f"  Page {page}/{max_pages}: {len(articles)} articles, new so far: {total_new}")

conn.commit()
conn.close()
print(f"\nDone! New: {total_new}")
