#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
烟台市福山区-建设项目环评审批爬虫
crawl_fushan_hjxxgk.py
列表: 主题分类list AJAX API (pageNo/pageSize/pageSize=100)
详情: div.text#zoom (正文), div.title h1 (标题)
附件: /api-gateway/jpaas-web-server/front/document/download?fileUrl=...
"""
import re, json, time, os, sys, urllib.parse, hashlib
import urllib.request, ssl, http.cookiejar
from datetime import datetime

CURRENT_DIR = os.path.dirname(os.path.abspath(__file__))
PARENT_DIR = os.path.dirname(CURRENT_DIR)
if PARENT_DIR not in sys.path:
    sys.path.insert(0, PARENT_DIR)

# 站点配置
BASE_URL = "http://www.ytfushan.gov.cn"
LIST_URL = BASE_URL + "/col/col44281/index.html?vc_xxgkarea=11370611004264003A&number=FSB300106&jh=263"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
SITE_NAME = "烟台市福山区-建设项目环评审批"
GROUP = "山东"
MAX_PAGES = 5
PAGE_SIZE = 100
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")

ctx = ssl._create_unverified_context()
cj = http.cookiejar.CookieJar()

def make_opener():
    https_handler = urllib.request.HTTPSHandler(context=ctx)
    opener = urllib.request.build_opener(https_handler, urllib.request.HTTPCookieProcessor(cj))
    opener.addheaders = [
        ('User-Agent', 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'),
        ('Accept', 'text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8'),
        ('Accept-Language', 'zh-CN,zh;q=0.9,en;q=0.8'),
    ]
    return opener

def init_session():
    """初始化session，获取cookie"""
    opener = make_opener()
    req = urllib.request.Request(LIST_URL)
    try:
        resp = opener.open(req, timeout=15)
        raw = resp.read()
    except:
        pass
    return opener

def fetch_list_page(opener, page_no):
    """通过API获取列表页数据"""
    search_params = {
        'title': '', 'createdate': '', 'depolytime': '', 'indexcode': '',
        'xxgkId': '', 'xxgkType': '', 'nodeId': '', 'isFindChild': ''
    }
    param_json = {
        'pageNo': page_no,
        'pageSize': PAGE_SIZE,
        'search': json.dumps(search_params, ensure_ascii=False)
    }
    params = {
        'parseType': 'bulidstatic',
        'unitType': 'ajax-xxgk',
        'webId': '148',
        'tplSetId': 'FkT6xljcB7CqYjedWPMTZ',
        'pageType': 'column',
        'tagId': '主题分类list',
        'editType': 'null',
        'pageId': '44281',
        'paramJson': json.dumps(param_json, ensure_ascii=False)
    }
    qs = urllib.parse.urlencode(params)
    url = f'{API_URL}?{qs}'
    req = urllib.request.Request(url, headers={
        'User-Agent': 'Mozilla/5.0',
        'Accept': 'application/json, text/plain, */*',
        'Referer': LIST_URL
    })
    resp = opener.open(req, timeout=20)
    raw = resp.read()
    import chardet
    enc = chardet.detect(raw)['encoding'] or 'utf-8'
    result = json.loads(raw.decode(enc, errors='replace'))
    if not result.get('success'):
        return [], 0, 0
    
    html = result['data']['html']
    count_m = re.search(r'count="(\d+)"', html)
    total_count = int(count_m.group(1)) if count_m else 0
    
    # 提取列表项 - 按行解析
    # 寻找所有行: <tr>...<a href="...">标题</a>...<td>日期</td>...</tr>
    items = []
    rows = re.findall(r'<tr[^>]*>(.*?)</tr>', html, re.S)
    for row in rows:
        # 提取链接和标题
        link_m = re.search(r'<a[^>]*href="([^"]*)"[^>]*>([^<]*)</a>', row)
        if not link_m:
            continue
        href = link_m.group(1).strip()
        title = link_m.group(2).strip()
        if not title or title in ('序号', '标题', '发布日期'):
            continue
        if not href.startswith('http'):
            href = urllib.parse.urljoin(BASE_URL, href)
        
        # 提取日期
        date_m = re.search(r'(\d{4}-\d{2}-\d{2})', row)
        publish_date = date_m.group(1) if date_m else ''
        
        items.append({
            'title': title,
            'url': href,
            'publish_date': publish_date
        })
    
    return items, total_count, len(items)

def fetch_detail(opener, url):
    """获取详情页内容"""
    req = urllib.request.Request(url)
    try:
        resp = opener.open(req, timeout=20)
        raw = resp.read()
    except Exception as e:
        return None, str(e)
    
    import chardet
    enc = chardet.detect(raw)['encoding'] or 'utf-8'
    html = raw.decode(enc, errors='replace')
    
    # 标题
    title = ''
    title_m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.S)
    if title_m:
        title = re.sub(r'<[^>]+>', '', title_m.group(1)).strip()
    if not title:
        title_m = re.search(r'<title>(.*?)</title>', html, re.S)
        if title_m:
            title = title_m.group(1).strip()
    
    # 发布日期
    publish_date = ''
    date_m = re.search(r'日期[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if date_m:
        publish_date = date_m.group(1)
    if not publish_date:
        date_m = re.search(r'(\d{4}-\d{2}-\d{2})\s+\d{2}:\d{2}', html)
        if date_m:
            publish_date = date_m.group(1)
    
    # 正文 - div.text#zoom
    content = ''
    text_m = re.search(r'<div[^>]*class="text"[^>]*id="zoom"[^>]*>(.*?)</div>\s*</div>', html, re.S)
    if not text_m:
        text_m = re.search(r'<div[^>]*class="text"[^>]*>(.*?)</div>\s*(?:</div>)?', html, re.S)
    if text_m:
        content_html = text_m.group(1)
        # 去除表格内的<p>和wzy_titlebox信息表，避免重复
        content_html_no_tables = re.sub(r'<table[^>]*wzy_titlebox[^>]*>.*?</table>', '', content_html, flags=re.S)
        content_html_no_tables = re.sub(r'<table[^>]*>.*?</table>', '', content_html_no_tables, flags=re.S)
        # 智能解析：保留<a>为[text](url)，<img>为![alt](src)
        paragraphs = []
        for p in re.findall(r'<p[^>]*>(.*?)</p>', content_html_no_tables, re.S):
            # 处理<a>标签
            p_text = re.sub(r'<a[^>]*href="([^"]*)"[^>]*>([^<]*)</a>', r'[\2](\1)', p)
            # 处理<img>标签
            p_text = re.sub(r'<img\s+[^>]*src="([^"]*)"[^>]*>', r'![](\1)', p_text)
            # 去掉剩余标签
            text = re.sub(r'<[^>]+>', '', p_text).strip()
            text = re.sub(r'\s+', ' ', text)
            text = re.sub(r'&nbsp;', ' ', text).strip()
            if text:
                paragraphs.append(text)
        if paragraphs:
            content = '\n\n'.join(paragraphs)
        else:
            content = re.sub(r'<[^>]+>', '', content_html).strip()
            content = re.sub(r'\s+', ' ', content).strip()
    
    # 表格 - 只从正文区提取（跳过信息表），保留 HTML
    tables = []
    scope_html = text_m.group(1) if text_m else html
    for tm in re.finditer(r'<table[^>]*>.*?</table>', scope_html, re.S):
        whole_table = tm.group(0)
        if 'wzy_titlebox' in whole_table:
            continue
        tables.append(whole_table)

    if tables:
        content += '\n\n' + '\n\n'.join(tables)
    
    # 附件 - 只从正文区提取
    attachments = []
    for m in re.finditer(r'<a[^>]*href="([^"]*)"[^>]*>([^<]*)</a>', scope_html, re.S):
        href = m.group(1).strip()
        text = m.group(2).strip()
        if '/document/download' in href or re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
            if not href.startswith('http'):
                href = urllib.parse.urljoin(BASE_URL, href)
            attachments.append({
                'name': text or href.split('/')[-1],
                'url': href
            })
            # 在正文末尾添加附件链接
            content += f'\n\n附件：<a href="{href}">{text}</a>'
    
    return {
        'title': title,
        'publish_date': publish_date,
        'content': content.strip(),
        'attachments': json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    }, None


def crawl():
    opener = init_session()
    print(f"[{SITE_NAME}] Session initialized")
    
    all_items = []
    total_api_count = 0
    
    for page in range(1, MAX_PAGES + 1):
        try:
            items, total_count, got = fetch_list_page(opener, page)
            total_api_count = total_count
            print(f"  Page {page}/{MAX_PAGES}: got {got} items (total in system: {total_count})")
            all_items.extend(items)
            time.sleep(0.3)
        except Exception as e:
            print(f"  Page {page}: ERROR - {e}")
            break
    
    print(f"\nTotal items from list: {len(all_items)}")
    
    # 过滤环评审批（col44281）文章
    hj_items = [it for it in all_items if '/col/col44281/' in it['url']]
    print(f"环评审批 items: {len(hj_items)}")
    
    saved = 0
    errors = 0
    for idx, item in enumerate(hj_items):
        try:
            detail, err = fetch_detail(opener, item['url'])
            if err:
                print(f"  [{idx+1}/{len(hj_items)}] ERROR fetching {item['title'][:30]}: {err}")
                errors += 1
                time.sleep(0.5)
                continue
            
            import sqlite3
            conn = sqlite3.connect(DB_PATH, timeout=60)
            c = conn.cursor()
            content_text = detail['content'][:50000] if detail['content'] else ''
            summary = re.sub(r'\s+', ' ', content_text[:200]).strip() or detail['title']
            c.execute('''INSERT OR IGNORE INTO gov_raw 
                (page_url, title, publish_date, site_name, group_name, summary, content, attachments, source_url)
                VALUES (?,?,?,?,?,?,?,?,?)''',
                (item['url'], detail['title'], detail['publish_date'],
                 SITE_NAME, GROUP, summary, content_text,
                 detail.get('attachments', '[]'), item['url']))
            n = c.rowcount
            conn.commit()
            conn.close()
            saved += 1
            if (idx + 1) % 10 == 0:
                print(f"  [{idx+1}/{len(hj_items)}] saved {saved}, errors {errors}")
            time.sleep(0.3)
        except Exception as e:
            print(f"  [{idx+1}/{len(hj_items)}] ERROR saving {item['title'][:30]}: {e}")
            errors += 1
            time.sleep(0.5)
    
    print(f"\n{'='*50}")
    print(f"[{SITE_NAME}] Done! Saved: {saved}, Errors: {errors}")
    print(f"{'='*50}")

if __name__ == '__main__':
    crawl()
