#!/usr/bin/env python3
"""
屯留区人民政府 - 环境保护领域 爬虫
http://www.tunliu.gov.cn/tlzw/zwgk/zfxxgkml/gzdt_203764/hjbh/
"""
import requests
import re
import sys
import json
import time
import sqlite3
import os
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
BASE_URL = "http://www.tunliu.gov.cn"
SITE_NAME = "屯留区人民政府"
COLUMN = "环境保护领域"
MAX_PAGES = 5
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TIMEOUT = 30
# Subsections: 行政许可(xzxk), 行政处罚(hj_xzcf1), 公共服务(ggfw)
SUBSECTIONS = ["xzxk", "hj_xzcf1", "ggfw"]

def get_page(url, session):
    for attempt in range(3):
        try:
            r = session.get(url, headers=HEADERS, timeout=TIMEOUT)
            r.encoding = 'utf-8'
            if r.status_code == 200:
                return r.text
        except Exception as e:
            if attempt < 2:
                time.sleep(3)
    return None

def parse_list(html, base_sub_url):
    """Parse list page from ul.main li items"""
    items = []
    soup = BeautifulSoup(html, 'html.parser')
    ul = soup.find('ul', class_='main')
    if not ul:
        return items
    for li in ul.find_all('li'):
        a = li.find('a', href=True)
        if not a:
            continue
        href = a['href']
        # Make absolute URL
        if href.startswith('./'):
            href = base_sub_url + href[2:]
        elif not href.startswith('http'):
            href = base_sub_url + href.lstrip('/')
        # Title
        title = a.get('title', '') or a.get_text(strip=True)
        # Date from span
        span = li.find('span')
        date = span.get_text(strip=True).replace('\xa0', '') if span else ''
        if href and title:
            items.append({'url': href, 'title': title, 'date': date})
    return items

def parse_detail(html, url):
    """Parse detail page"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # Title from meta
    title = ''
    meta_title = soup.find('meta', attrs={'name': 'ArticleTitle'})
    if meta_title and meta_title.get('content'):
        title = meta_title['content'].strip()
    if not title:
        title_el = soup.find('title')
        if title_el:
            t = title_el.get_text(strip=True)
            title = t.replace('_屯留区人民政府', '').replace('_长治市屯留区人民政府', '')
    
    # Date from meta
    date = ''
    meta_date = soup.find('meta', attrs={'name': 'PubDate'})
    if meta_date and meta_date.get('content'):
        date = meta_date['content'].strip()
    
    # Source
    source = ''
    meta_source = soup.find('meta', attrs={'name': 'ContentSource'})
    if meta_source and meta_source.get('content'):
        source = meta_source['content'].strip()
    
    # Content from div.article-con.TRS_Editor
    # Find the innermost editor div to avoid nested duplication
    content_div = soup.find('div', class_='article-con') or soup.find('div', class_='TRS_Editor')
    if content_div:
        # Prefer the inner TRS editor view if it exists
        inner = content_div.find('div', class_='TRS_UEDITOR')
        if inner:
            content_div = inner
    content_parts = []
    attachments = []
    total_text_len = 0
    found_images = False
    seen_texts = set()  # dedup
    
    if content_div:
        for img in content_div.find_all('img'):
            src = img.get('src', '')
            if src and not src.startswith('http'):
                src = BASE_URL + src
            alt = img.get('alt', '')
            found_images = True
            content_parts.append(f'![{alt}]({src})')
        
        # Walk direct children of content_div for structured content
        for child in list(content_div.children):
            if child.name == 'table':
                content_parts.append(str(child))
                total_text_len += len(child.get_text(strip=True))
            elif child.name == 'p':
                text = child.get_text(strip=True)
                if text and text not in seen_texts:
                    seen_texts.add(text)
                    content_parts.append(text)
                    total_text_len += len(text)
            elif child.name == 'div':
                # Check if this div contains a table
                tbl = child.find('table')
                if tbl:
                    content_parts.append(str(tbl))
                    total_text_len += len(tbl.get_text(strip=True))
                # Also get text from non-table parts of this div
                for p in child.find_all('p', recursive=False):
                    text = p.get_text(strip=True)
                    if text and text not in seen_texts and not p.find_parent('td'):
                        seen_texts.add(text)
                        content_parts.append(text)
                        total_text_len += len(text)
                # Check for images in inner divs
                for img in child.find_all('img'):
                    src = img.get('src', '')
                    if src and not src.startswith('http'):
                        src = BASE_URL + src
                    alt = img.get('alt', '')
                    found_images = True
                    if f'![{alt}]({src})' not in content_parts:
                        content_parts.append(f'![{alt}]({src})')
        
        for a in content_div.find_all('a', href=True):
            href = a['href']
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                if not href.startswith('http'):
                    href = BASE_URL + href
                name = a.get_text(strip=True) or href.split('/')[-1]
                attachments.append({'name': name, 'url': href})
    
    # Image-only / low-text fallback
    if total_text_len < 20:
        fallback = f"[{title}]({url})"
        if found_images:
            content_parts = [fallback] + [p for p in content_parts if p.startswith('![')]
        else:
            content_parts = [fallback]
            pdf_links = [a for a in attachments if re.search(r'\.pdf$', a['url'], re.I)]
            if pdf_links:
                for pdf in pdf_links:
                    content_parts.append(f"[PDF附件: {pdf['name']}]({pdf['url']})")
    
    content = '\n\n'.join(content_parts)
    summary = re.sub(r'\s+', ' ', content[:200]).strip() if content else title
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    
    return title, date, content, summary, attachments_json

def save_to_db(items, conn):
    cursor = conn.cursor()
    count = 0
    for item in items:
        publish_date = item.get('date', '')
        if publish_date:
            try:
                dt = datetime.strptime(publish_date[:10], '%Y-%m-%d')
                if dt > datetime.now().replace(hour=0, minute=0, second=0, microsecond=0) + __import__('datetime').timedelta(days=30):
                    publish_date = ''
            except:
                publish_date = ''
        
        cursor.execute("""
            INSERT OR REPLACE INTO gov_raw (page_url, site_name, title, publish_date, summary, content, attachments, status, group_name, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, 'active', ?, 'crawl_tunliu.py')
        """, (
            item['url'], SITE_NAME, item['title'],
            publish_date, item.get('summary', ''),
            item.get('content', ''), item.get('attachments', '[]'),
            SITE_NAME
        ))
        count += 1
    conn.commit()
    return count

def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument('--max-pages', type=int, default=MAX_PAGES)
    parser.add_argument('--subs', nargs='+', default=SUBSECTIONS,
                        help='subsections to crawl')
    args = parser.parse_args()
    max_pages = args.max_pages
    subsections = args.subs
    
    BASE_SUB = "http://www.tunliu.gov.cn/tlzw/zwgk/zfxxgkml/gzdt_203764/hjbh"
    session = requests.Session()
    all_items = []
    
    for sub in subsections:
        print(f"\n=== 子栏目: {sub} ===", flush=True)
        sub_base = f"{BASE_SUB}/{sub}/"
        
        for page_num in range(1, max_pages + 1):
            if page_num == 1:
                url = sub_base
            else:
                url = f"{sub_base}index_{page_num-1}.html"  # 0-based pagination
            
            print(f"[列表页] {sub} 第{page_num}页: {url}", flush=True)
            html = get_page(url, session)
            if not html:
                print(f"  -> 获取失败", flush=True)
                break
            
            items = parse_list(html, sub_base)
            if not items:
                print(f"  -> 无数据，停止", flush=True)
                break
            
            print(f"  -> {len(items)}条", flush=True)
            
            # Detect total pages from first page
            if page_num == 1:
                m = re.search(r'countPage\s*=\s*(\d+)', html)
                if m:
                    page_count = int(m.group(1))
                    print(f"  (共{page_count}页)", flush=True)
            
            all_items.extend(items)
            if page_num >= max_pages:
                break
    
    if not all_items:
        print("无数据")
        return
    
    print(f"\n列表共 {len(all_items)} 条，开始获取详情...", flush=True)
    
    detailed = []
    for i, item in enumerate(all_items):
        print(f"  [{i+1}/{len(all_items)}] {item['title'][:30]}...", flush=True)
        html = get_page(item['url'], session)
        if not html:
            print(f"    -> 详情获取失败，跳过", flush=True)
            continue
        
        title, date, content, summary, attachments = parse_detail(html, item['url'])
        item['title'] = title or item['title']
        item['date'] = date or item['date']
        item['content'] = content
        item['summary'] = summary
        item['attachments'] = attachments
        detailed.append(item)
        
        if (i + 1) % 10 == 0:
            time.sleep(1)
    
    conn = sqlite3.connect(DB_PATH, timeout=60)
    count = save_to_db(detailed, conn)
    conn.close()
    
    errors = len(all_items) - len(detailed)
    print(f"\n完成: {count}条, 异常: {errors}")

if __name__ == '__main__':
    main()
