#!/usr/bin/env python3
"""龙港市人民政府 - 行政许可（生态环境领域）爬虫
站点: www.zjlg.gov.cn
栏目: /col/col1229549231/index.html (行政许可 - 生态环境基层政务公开)
数据通过JCMS API加载
"""

import requests
import sqlite3
import re
import sys
import os
import json
from datetime import datetime
from bs4 import BeautifulSoup

BASE_URL = "http://www.zjlg.gov.cn"
API_URL = BASE_URL + "/api-gateway/jpaas-publish-server/front/page/build/unit"
DB_PATH = "/root/search.db"
SITE_NAME = "www.zjlg.gov.cn-xzxk"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
    "Accept": "*/*",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "http://www.zjlg.gov.cn/col/col1229549231/index.html",
}

API_PARAMS = {
    "parseType": "bulidstatic",
    "webId": "2016",
    "tplSetId": "tfABXwWYi3U9Vix6VgDl6",
    "pageType": "column",
    "tagId": "重点领域右列表",
    "editType": "null",
    "pageId": "1229549231",
}

def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    return conn

def fetch_api(page_no=1, page_size=20):
    """调用JCMS API获取列表数据"""
    params = dict(API_PARAMS)
    params["paramJson"] = json.dumps({"pageNo": page_no, "pageSize": page_size}, ensure_ascii=False)
    try:
        resp = requests.get(API_URL, headers=HEADERS, params=params, timeout=30, verify=False)
        if resp.status_code == 200:
            data = resp.json()
            if data.get('success'):
                return data['data'].get('html', '')
        return None
    except Exception as e:
        print(f"  [ERROR] API call failed: {e}", file=sys.stderr)
        return None

def fetch_page(url, timeout=30):
    try:
        resp = requests.get(url, headers={
            "User-Agent": HEADERS["User-Agent"],
            "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
            "Accept-Language": "zh-CN,zh;q=0.9",
        }, timeout=timeout, verify=False)
        resp.encoding = 'utf-8'
        if resp.status_code == 200:
            return resp.text
        return None
    except Exception as e:
        print(f"  [ERROR] fetch failed: {e}", file=sys.stderr)
        return None

def parse_list(html_text):
    """解析API返回的HTML列表"""
    items = []
    soup = BeautifulSoup(html_text, 'html.parser')
    for li in soup.find_all('li'):
        a = li.find('a')
        if not a or not a.get('href'):
            continue
        href = a['href'].strip()
        if not href.endswith('.html'):
            continue
        if href.startswith('/'):
            href = BASE_URL + href
        elif not href.startswith('http'):
            href = BASE_URL + '/' + href
        # 标题
        title = a.get_text(strip=True)
        if not title:
            continue
        # 日期
        date_span = li.find('span')
        pub_date = date_span.get_text(strip=True) if date_span else ''
        date_match = re.search(r'(\d{4}-\d{2}-\d{2})', pub_date)
        pub_date = date_match.group(1) if date_match else ''
        items.append({'url': href, 'title': title, 'pub_date': pub_date})
    return items

def parse_detail(html, url):
    """解析详情页"""
    soup = BeautifulSoup(html, 'html.parser')
    
    # 标题 - 从div.tt取发布日期前面的标题
    title = ''
    title_tag = soup.find('title')
    if title_tag:
        t = title_tag.get_text(strip=True)
        t = re.sub(r'\s*[-|]\s*.*$', '', t).strip()
        title = t
    
    # 日期
    pub_date = ''
    for m in re.finditer(r'(\d{4}-\d{2}-\d{2})', html):
        pub_date = m.group(1)
        break
    
    # 正文 - div#zoom
    content_div = soup.find('div', id='zoom')
    attachments = []
    content = ''
    
    if content_div:
        for tag in content_div.find_all(['script', 'style']):
            tag.decompose()
        
        # 提取表格HTML
        tables_html = ''
        seen_texts = set()
        for table in content_div.find_all('table'):
            rows = table.find_all('tr')
            has_data = any(len(row.find_all('td')) >= 3 for row in rows) if rows else False
            if not has_data:
                continue
            txt = table.get_text(strip=True)
            if txt in seen_texts:
                continue
            seen_texts.add(txt)
            tables_html += str(table) + '\n\n'
        
        content_div_clean = BeautifulSoup(str(content_div), 'html.parser')
        for tag in content_div_clean.find_all('table'):
            tag.decompose()
        
        content = content_div_clean.get_text(separator='\n\n', strip=True)
        content = re.sub(r'\n{4,}', '\n\n', content)
        if tables_html:
            content += '\n\n[表格]\n' + tables_html
    
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>'
    
    return title, content, pub_date, attachments


def save_article(conn, page_url, title, summary_text, publish_date, attachments, site_name):
    attachments_json = json.dumps(attachments, ensure_ascii=False) if attachments else '[]'
    now = datetime.now().strftime('%Y-%m-%d %H:%M:%S')
    level = abs(hash(page_url)) % 10 + 1
    conn.execute("""
        INSERT OR REPLACE INTO gov_raw (page_url, site_name, source_url, title, publish_date, summary, content, attachments, date_rank, status, category, visits, script_name) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, 'published', '行政许可', 0, 'crawl_zjlg_xzxk.py')
    """, (page_url, site_name, page_url, title, publish_date, summary_text, summary_text, attachments_json, level))
    conn.commit()


def crawl(max_pages=5, incremental=False):
    conn = init_db()
    total = 0
    errors = 0
    seen_urls = set()

    for page in range(max_pages):
        print(f"[PAGE {page+1}] API pageNo={page+1}")
        html_text = fetch_api(page_no=page+1, page_size=20)
        if not html_text or len(html_text) < 100:
            print(f"  [WARN] No more data, stopping")
            break
        
        items = parse_list(html_text)
        unique_items = [it for it in items if it['url'] not in seen_urls]
        for it in items:
            seen_urls.add(it['url'])
        print(f"  Found {len(items)} articles ({len(unique_items)} new)")
        
        if not unique_items:
            break
        
        for item in unique_items:
            url = item['url']
            title = item['title']
            list_date = item['pub_date']
            
            if incremental:
                existing = conn.execute(
                    "SELECT page_url FROM gov_raw WHERE page_url = ?", (url,)
                ).fetchone()
                if existing:
                    continue
            
            print(f"  [FETCH] {title[:50]}...")
            detail_html = fetch_page(url)
            if not detail_html:
                errors += 1
                print(f"  [ERROR] detail page unreachable: {url}")
                continue
            
            detail_title, content, pub_date, attachments = parse_detail(detail_html, url)
            if not pub_date and list_date:
                pub_date = list_date
            
            if detail_title:
                save_article(conn, url, detail_title, content, pub_date, attachments, SITE_NAME)
                total += 1
                print(f"    ✓ saved [{pub_date}] {detail_title[:50]}...")
            else:
                print(f"    ✗ empty title, skipped")
    
    conn.close()
    print(f"\n[DONE] Total: {total} articles, Errors: {errors}")


if __name__ == '__main__':
    import argparse
    parser = argparse.ArgumentParser(description='龙港市生态环境行政许可爬虫')
    parser.add_argument('--max-pages', type=int, default=5, help='最大爬取页数')
    parser.add_argument('--pages', type=int, default=None, help='最大爬取页数(兼容调度器)')
    parser.add_argument('--incremental', action='store_true', help='增量模式')
    args = parser.parse_args()
    mp = args.pages if args.pages else args.max_pages
    crawl(max_pages=mp, incremental=args.incremental)
