#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
沂源县人民政府 - 环评审批信息
路径: 首页 > 政务公开 > 政府信息公开 > 环境保护 > 建设项目影响评价 > 环评审批信息
CMS: Hanweb 政府信息公开平台
"""

import requests
import re
import sys
import json
import os
import subprocess
from bs4 import BeautifulSoup

BASE_URL = "http://www.yiyuan.gov.cn"
LIST_URL = "http://www.yiyuan.gov.cn/gongkai/channel_c_5f9f696d489871774b4da7ea_n_1605850085.8145/"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}
DB_PATH = "/mnt/data/search.db"

def fetch(url, encoding='utf-8'):
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = encoding
        r.raise_for_status()
        return r.text
    except Exception as e:
        print(f"FETCH ERROR {url}: {e}", file=sys.stderr)
        return None

def parse_list(html):
    """Parse 环评审批信息 list page - extract actual EIA article links"""
    soup = BeautifulSoup(html, 'html.parser')
    articles = []
    
    for a in soup.find_all('a', href=True):
        href = a['href']
        txt = a.get_text(strip=True)
        
        # Real article links: /doc_ in href, and title has date suffix (YYYY-MM-DD)
        if '/doc_' in href and len(txt) > 10:
            # Filter: only keep EIA-related articles (contain 环评/审批/受理/公示 in title)
            # and NOT department listing links (those without dates)
            if re.search(r'(环评|审批|受理|公示|公告)', txt) and re.search(r'\d{4}-\d{2}-\d{2}', txt):
                title = re.sub(r'\s*\d{4}-\d{2}-\d{2}\s*$', '', txt).strip()
                if href.startswith('http'):
                    full_url = href
                elif href.startswith('/'):
                    full_url = BASE_URL + href
                else:
                    full_url = BASE_URL + '/' + href
                
                # Deduplicate
                if not any(a['url'] == full_url for a in articles):
                    articles.append({'title': title, 'url': full_url})
    
    print(f"List: {len(articles)} EIA articles found")
    return articles

def parse_detail(html, url):
    soup = BeautifulSoup(html, 'html.parser')
    
    title = ""
    mt = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if mt and mt.get('content'):
        title = mt['content'].strip()
    if not title:
        h1 = soup.find('h1')
        if h1:
            title = h1.get_text(strip=True)
    
    pub_date = ""
    mp = soup.find('meta', attrs={'name': 'PubDate'})
    if mp and mp.get('content'):
        pub_date = mp['content'].strip()[:10]
    
    content = ""
    dc = soup.find('div', id='details-content')
    if dc:
        # Preserve tables as HTML, extract text for non-table elements
        parts = []
        for child in dc.children:
            if child.name:
                # Check if this element IS a table or contains a table
                if child.name == 'table' or child.find_all('table'):
                    parts.append(str(child))
                else:
                    txt = child.get_text('', strip=True)
                    if txt:
                        parts.append(txt)
            elif child.string and child.string.strip():
                parts.append(child.string.strip())
        content = '\n\n'.join(parts)
    
    # Attachments
    attachments = []
    if dc:
        for a in dc.find_all('a', href=True):
            h = a['href']
            atxt = a.get_text(strip=True)
            if any(h.lower().endswith(ext) for ext in ['.pdf', '.doc', '.docx', '.xls', '.xlsx', '.zip', '.rar']):
                if h.startswith('/'):
                    h = BASE_URL + h
                attachments.append(f'<a href="{h}">{atxt}</a>')
    
    if attachments:
        content += '\n\n附件：\n' + '\n'.join(attachments)
    
    return {
        'title': title or url,
        'publish_date': pub_date,
        'content': content,
        'source_url': url,
    }

def crawl():
    html = fetch(LIST_URL)
    if not html:
        print("ERROR: Failed to fetch list page")
        return []
    
    articles = parse_list(html)
    results = []
    
    for i, art in enumerate(articles):
        print(f"[{i+1}/{len(articles)}] {art['title'][:50]}...")
        html_d = fetch(art['url'])
        if not html_d:
            continue
        detail = parse_detail(html_d, art['url'])
        if not detail['title']:
            detail['title'] = art['title']
        results.append(detail)
    
    return results

def push_to_searchdb(results):
    import sqlite3
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    c = conn.cursor()
    count = 0
    
    for r in results:
        try:
            c.execute('''INSERT OR IGNORE INTO gov_raw 
                (title, content, publish_date, page_url, site_name, group_name, industry, summary)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?)''', (
                r['title'], r['content'], r.get('publish_date', ''),
                r['source_url'], '沂源县人民政府', '山东省', '环评公示', r['title'][:200]
            ))
            if c.rowcount > 0:
                count += 1
        except Exception as e:
            print(f"  DB ERROR: {e}", file=sys.stderr)
    
    conn.commit()
    conn.close()
    print(f"DB: {count} new records inserted")
    
    # FTS sync
    if count > 0:
        print("Syncing FTS...")
        conn2 = sqlite3.connect(DB_PATH, timeout=60)
        c2 = conn2.cursor()
        unsynced = c2.execute("""
            SELECT r.id, r.title, r.site_name
            FROM gov_raw r
            LEFT JOIN gov_search s ON r.id = s.rowid
            WHERE r.site_name='沂源县人民政府' AND s.rowid IS NULL
        """).fetchall()
        conn2.close()
        
        for us in unsynced:
            rowid = us[0]
            title = us[1].replace("'", "''")
            site_name = us[2].replace("'", "''")
            sql = f"INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary) VALUES ({rowid}, '{title}', '{site_name}', '{title}');"
            subprocess.run(['sqlite3', "-cmd", ".timeout 60000", DB_PATH], input=sql, capture_output=True, text=True, timeout=10)
        
        print(f"FTS synced: {len(unsynced)} rows")
    
    return count

def main():
    print(f"=== 沂源县-环评审批信息 ===")
    print(f"List URL: {LIST_URL}")
    
    results = crawl()
    print(f"\nTotal crawled: {len(results)}")
    
    if results:
        new = push_to_searchdb(results)
        print(f"New to DB: {new}")
    
    print("Done.")

if __name__ == '__main__':
    main()
