#!/usr/bin/env python3
"""
天津经济技术开发区 - 行政许可服务事项 (teda.gov.cn)
https://www.teda.gov.cn/bmztc/channels/1171.html?legal=行政许可、服务事项
列表: API /api/stl/actions/dynamic (POST JSON, value从页面提取)
详情: 静态HTML, div.article_con
"""
import sys, os, re, time, json
from datetime import datetime, timezone, timedelta
import requests, urllib3
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb
urllib3.disable_warnings(urllib3.exceptions.InsecureRequestWarning)

SITE_NAME = "天津经开区-行政许可"
BASE_URL = "https://www.teda.gov.cn"
LIST_URL = "https://www.teda.gov.cn/bmztc/channels/1171.html"
THREE_YEARS_AGO = (datetime.now(timezone.utc) - timedelta(days=365*3)).strftime("%Y-%m-%d")
MAX_PAGES = 5
HEADERS = {"User-Agent": "Mozilla/5.0", "X-Requested-With": "XMLHttpRequest"}

def fetch(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=20, verify=False)
        r.encoding = 'utf-8'
        return r.text
    except: return None

def get_encrypted_value(html):
    """从页面提取加密的API value"""
    m = re.search(r"stlClient\.post\('(/api/stl/actions/dynamic\?)' \+ StlClient\.getQueryString\(\), \{\s*value: '([^']+)'", html)
    if m:
        return m.group(1), m.group(2)
    return None, None

def parse_list_from_api(page_num=1):
    """通过API获取列表页数据"""
    items = []
    try:
        legal_value = "行政许可、服务事项"
        legal_enc = "%E8%A1%8C%E6%94%BF%E8%AE%B8%E5%8F%AF%E3%80%81%E6%9C%8D%E5%8A%A1%E4%BA%8B%E9%A1%B9"
        # 第一步：获取页面提取加密value
        resp = requests.get(f"{LIST_URL}?legal={legal_enc}&page={page_num}",
                          headers=HEADERS, timeout=20, verify=False)
        resp.encoding = 'utf-8'
        html = resp.text
        
        api_path, enc_value = get_encrypted_value(html)
        if not api_path or not enc_value:
            return items
        
        value = enc_value.replace('0slash0', '/').replace('0add0', '+').replace('0equals0', '=')
        
        # 第二步：调用API，必须带上legal参数
        api_resp = requests.post(BASE_URL + api_path,
            params={'legal': legal_value, 'page': str(page_num)},
            json={'value': value, 'page': page_num},
            headers={**HEADERS, 'Content-Type': 'application/json',
                     'Referer': f'{LIST_URL}?legal={legal_enc}&page={page_num}'},
            timeout=20, verify=False
        )
        
        if api_resp.status_code == 200:
            data = api_resp.json()
            list_html = data.get('html', '')
            # 解析列表项 - 只取含项目链接的条目
            pattern = r'<a[^>]*href="(/bmztc/contents/\d+/\d+\.html)"[^>]*>(.*?)</a>'
            for m in re.finditer(pattern, list_html, re.DOTALL):
                href = m.group(1)
                title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
                import html as _html
                title = _html.unescape(title)
                title = re.sub(r'^\s*[•·\s]+\s*', '', title)
                if title and len(title) > 5 and '项目' in title:
                    full_url = BASE_URL + href
                    items.append((title, full_url))
    except Exception as e:
        print(f"  ⚠ API Error: {e}")
    return items

def extract_nested(html, tag_start):
    """找到匹配的闭合标签，支持嵌套div"""
    depth = 0
    i = tag_start
    n = len(html)
    while i < n:
        if html[i:i+4] == '<!--':
            end = html.find('-->', i+4)
            if end > i: i = end + 3
            else: i += 1
            continue
        if html[i:i+6] == '</div>':
            if depth == 0:
                return html[tag_start:i]
            depth -= 1
            i += 6
            continue
        if html[i:i+4] == '<div' and html[i+4] in (' ', '>', '\n', '\t', '\r'):
            depth += 1
            i += 4
            continue
        if html[i:i+8] == '</section>':
            if depth == 0:
                return html[tag_start:i]
            depth -= 1
            i += 8
            continue
        if html[i:i+8] == '<section' and html[i+8] in (' ', '>', '\n', '\t', '\r'):
            depth += 1
            i += 8
            continue
        i += 1
    return html[tag_start:]

def fetch_detail(url):
    html = fetch(url)
    if not html: return None, None, None
    result = {}
    m = re.search(r'<h1[^>]*>(.*?)</h1>', html, re.DOTALL)
    if m: result["title"] = re.sub(r"<[^>]+>", "", m.group(1)).strip()
    m = re.search(r'发布日期[：:]\s*(\d{4}-\d{1,2}-\d{1,2})', html)
    if m: result["publish_date"] = m.group(1)
    m = re.search(r'class="article_con"[^>]*>', html)
    if m:
        content = extract_nested(html, m.end()).strip()
        content = re.sub(r"<script[^>]*>.*?</script>", "", content, flags=re.DOTALL|re.I)
        content = re.sub(r"<style[^>]*>.*?</style>", "", content, flags=re.DOTALL|re.I)
        if len(content) > 100:
            result["content"] = content
    if not result.get("content"):
        for cls in ['content', 'article', 'text', 'main']:
            m = re.search(r'<(div|section)[^>]*class="' + cls + r'[^"]*"[^>]*>(.*?)</\1>', html, re.DOTALL)
            if m and len(m.group(2)) > 100:
                result["content"] = m.group(2).strip()
                break
    return result.get("title"), result.get("content"), result.get("publish_date")

def main():
    print(f"\n{'='*50}\n🏠 {SITE_NAME}\n{'='*50}")
    all_list = []
    for page in range(1, MAX_PAGES+1):
        print(f"\n📄 第 {page} 页...", end=" ", flush=True)
        items = parse_list_from_api(page)
        if not items: print("0 条"); break
        print(f"✅ {len(items)} 条")
        all_list.extend(items)
    print(f"\n📊 列表总计: {len(all_list)} 条")
    all_items, seen = [], set()
    for i, (title, url) in enumerate(all_list):
        if "项目" not in title: continue
        if url in seen: continue
        seen.add(url)
        print(f"  [{i+1}/{len(all_list)}] {title[:50]}...", end=" ", flush=True)
        dt, content, date = fetch_detail(url)
        if dt: title = dt
        summary = re.sub(r"<[^>]+>", " ", content or "").strip()[:300]
        summary = re.sub(r"\s+", " ", summary)
        all_items.append({
            "site_name": SITE_NAME, "title": title, "url": url,
            "content": content or "", "pub_date": date or "",
            "summary": summary, "tags": SITE_NAME,
        })
        print("✅")
        time.sleep(0.3)
    if all_items:
        push_to_searchdb(all_items, "teda_hpgs")
    print(f"\n✅ 完成! 共 {len(all_items)} 条")

if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"⏱ 耗时: {time.time()-t0:.1f}s")
