#!/usr/bin/env python3
"""
港城产业园区-信息公示 爬虫
站点: www.bhna.gov.cn/bhna/c104864/
列表: 15页, 每页20条, 共295条
正文: <div id="zoom">
"""
import sys, os, re, time, json, sqlite3, urllib.request
from datetime import datetime, timedelta
import os

SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")
BASE_URL = "http://www.bhna.gov.cn/bhna/c104864"
SITE_NAME = "港城产业园区-信息公示"
THREE_YEARS_AGO = datetime.now() - timedelta(days=3*365)
THREE_YEAR_LIMIT = True
MAX_PAGES = 5

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def save_to_db(records):
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    try:
        for r in records:
            db.execute("INSERT OR IGNORE INTO gov_raw(title, content, publish_date, source_url, page_url, site_name) VALUES (?,?,?,?,?,?)",
                       (r['title'], r['content'], r['publish_date'], r['url'], r['url'], SITE_NAME))
        db.commit()
    finally:
        db.close()

def fetch_list_page(page):
    if page == 1:
        url = f"{BASE_URL}/listDisplaySelf.shtml"
    else:
        url = f"{BASE_URL}/listDisplaySelf_{page}.shtml"
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        return html
    except Exception as e:
        print(f"  ⚠ 第{page}页失败: {e}")
        return ""

def parse_list(html):
    items = []
    # 找 wz_list 区域
    m = re.search(r'wz_list">(.*?)</ul>', html, re.DOTALL)
    if not m:
        return items
    content = m.group(1)
    lis = re.findall(r'<li>(.*?)</li>', content, re.DOTALL)
    for li in lis:
        a = re.search(r'href=["\']([^"\']+)["\'][^>]*>([^<]+)</a>', li)
        if not a:
            continue
        url = a.group(1).strip()
        title = a.group(2).strip()
        date_span = re.search(r'<font[^>]*>\s*(\d{4}-\d{2}-\d{2})</font>', li)
        date_str = date_span.group(1) if date_span else ""
        items.append((url, title, date_str))
    return items

def clean_zoom_html(content, base_url):
    """清洗 zoom 正文：剥 p/span 属性、img/a 相对路径绝对化"""
    try:
        from bs4 import BeautifulSoup
        from urllib.parse import urljoin
    except ImportError:
        return content
    soup = BeautifulSoup(content, "html.parser")
    for tag in soup.find_all(['p', 'span', 'div', 'font']):
        for a in list(tag.attrs):
            if a in ('class', 'style', 'id', 'align', 'border'):
                del tag[a]
    for img in soup.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http://', 'https://', 'data:')):
            img['src'] = urljoin(base_url, src)
    for a in soup.find_all('a', href=True):
        href = a.get('href', '')
        if href and not href.startswith(('http://', 'https://', 'javascript:', '#')):
            a['href'] = urljoin(base_url, href)
    return str(soup).strip()

def fetch_detail(url):
    full_url = url if url.startswith('http') else f"http://www.bhna.gov.cn{url}"
    req = urllib.request.Request(full_url, headers=HEADERS)
    try:
        resp = urllib.request.urlopen(req, timeout=30)
        html = resp.read().decode('utf-8', errors='replace')
        # 提取 zoom 正文 + 完整标题(h1优先)
        content = ""
        m = re.search(r'<div[^>]*id=["\']zoom["\'][^>]*>(.*?)</div>', html, re.DOTALL)
        if m:
            content = clean_zoom_html(m.group(1).strip(), full_url)
        title = ""
        h1 = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
        if h1:
            title = re.sub(r'<[^>]+>', '', h1.group(1)).strip()
        if not title:
            tt = re.search(r'<title>(.*?)</title>', html, re.DOTALL)
            if tt:
                title = re.sub(r'<[^>]+>', '', tt.group(1)).strip()
                title = re.sub(r'\s*[-_—|]\s*[^-_—|]*$', '', title).strip()
        return content, title
    except Exception as e:
        print(f"  ⚠ 详情失败: {url}: {e}")
        return "", ""

def main():
    print(f"{'='*50}")
    print(f"{SITE_NAME}")
    print(f"{'='*50}")
    
    all_articles = []
    all_urls = set()
    
    for page in range(1, MAX_PAGES + 1):
        html = fetch_list_page(page)
        if not html:
            continue
        items = parse_list(html)
        print(f"  📄 第{page}页: {len(items)}条", end="")
        for url, title, date_str in items:
            if url in all_urls:
                continue
            all_urls.add(url)
            if THREE_YEAR_LIMIT and date_str:
                try:
                    pub_date = datetime.strptime(date_str, '%Y-%m-%d')
                    if pub_date < THREE_YEARS_AGO:
                        continue
                except:
                    pass
            all_articles.append({'url': url, 'title': title, 'publish_date': date_str})
        print(f" → 累计{len(all_articles)}条")
    
    print(f"\n📊 列表总计: {len(all_articles)} 条")
    print(f"\n{'='*50}")
    print("抓取详情页...")
    
    batch = []
    total_saved = 0
    
    for i, art in enumerate(all_articles):
        title = art['title']
        print(f"  [{i+1}/{len(all_articles)}] {title[:50]}...", end=" ")
        
        content, detail_title = fetch_detail(art['url'])
        if content:
            final_title = detail_title or title
            final_title = re.sub(r'[\.]{3,}|…$', '', final_title).strip()
            batch.append({
                'title': final_title,
                'content': content,
                'publish_date': art['publish_date'],
                'url': f"http://www.bhna.gov.cn{art['url']}"
            })
            print(f"✅ ({len(content)}字)")
            if len(batch) >= 10:
                save_to_db(batch)
                total_saved += len(batch)
                batch = []
        else:
            print("⚠ 无正文")
    
    if batch:
        save_to_db(batch)
        total_saved += len(batch)
    
    print(f"\n{'='*50}")
    print(f"✅ 完成! 入库: {total_saved}条")
    
    if total_saved > 0:
        db = sqlite3.connect(SEARCH_DB, timeout=60)
        try:
            db.execute("INSERT INTO gov_search(rowid, title, site_name, summary) SELECT rowid, title, site_name, substr(content,1,200) FROM gov_raw WHERE site_name=? AND rowid NOT IN (SELECT rowid FROM gov_search WHERE site_name=?)", (SITE_NAME, SITE_NAME))
            db.commit()
            print("   FTS已更新")
        except Exception as e:
            print(f"   FTS更新: {e}")
        finally:
            db.close()

if __name__ == '__main__':
    main()
