#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
枣庄市市中区人民政府 - 生态环境分局 爬虫
URL: http://www.zzszq.gov.cn/zw/gkml/qzbm/sthjfj/
CMS: TRS 政府信息公开系统 (searPageNewShiZhong.jsp)
列表: iframe → searPageNewShiZhong.jsp?siteid=43&classinfoid=3441&channelid=12817
详情: <meta name="ArticleTitle"/> + <meta name="PubDate"/> + div.zwnr div.TRS_UEDITOR
"""
import re, sys, json, subprocess, time, os
from urllib.parse import urljoin
from bs4 import BeautifulSoup

# ── 配置 ──
SITE_NAME = "枣庄市市中区人民政府-生态环境分局"
GROUP = "山东省"
DB_PATH = "/mnt/data/search.db"
CHANNELID = "12817"

# 子分类 (与sthjfj相关的)
SUBCATEGORIES = {
    "3472": "工作信息",
    "3477": "行政权力运行公开",
    "3534": "重点领域信息公开",
    "3638": "政务公开保障机制",
}

LIST_URL = "http://www.zzszq.gov.cn/govsearch/searPageNewShiZhong.jsp"
MAX_PAGES = int(sys.argv[sys.argv.index("--pages")+1]) if "--pages" in sys.argv else 5

def fetch_list(cid, page=1):
    data = f"siteid=43&classinfoid={cid}&channelid={CHANNELID}&page={page}&pubURL=&indexPa=&schn=&curpos=&sinfo=&surl="
    r = subprocess.run(
        ["curl", "-s", "--max-time", "15", LIST_URL, "-d", data],
        capture_output=True, text=True
    )
    return r.stdout

def parse_items(html, base_url="http://www.zzszq.gov.cn"):
    items = []
    for m in re.finditer(
        r'<a[^>]*href="([^"]*)"[^>]*title="([^"]*)"[^>]*>\s*(?:<[^>]*>)*\s*([^<]*)',
        html
    ):
        url = m.group(1)
        title = m.group(2) or m.group(3).strip()
        rest = html[m.end():m.end()+200]
        dm = re.search(r'<b>\s*(\d{4}年\d{1,2}月\d{1,2}日)\s*</b>', rest)
        date = dm.group(1) if dm else ""
        
        if not url.startswith("http"):
            url = urljoin(base_url, url)
        if "sthjfj" in url and "t20" in url:
            items.append({"url": url, "title": title, "date": date})
    
    return items

def fetch_detail(url):
    r = subprocess.run(
        ["curl", "-s", "--max-time", "15", url],
        capture_output=True, text=True
    )
    return r.stdout

def extract_detail(html, url):
    soup = BeautifulSoup(html, 'html.parser')
    
    title_meta = soup.find('meta', attrs={"name": "ArticleTitle"})
    title = title_meta.get("content", "").strip() if title_meta else ""
    if not title or title == "枣庄市市中区人民政府":
        ttag = soup.find('title')
        if ttag:
            t = ttag.get_text(strip=True)
            title = t.replace("枣庄市市中区人民政府--", "").replace("枣庄市市中区人民政府", "").strip("-- ").strip()
    
    date_meta = soup.find('meta', attrs={"name": "PubDate"})
    date = date_meta.get("content", "").strip() if date_meta else ""
    
    content_html = ""
    content_div = soup.select_one('div.zwnr div.TRS_UEDITOR, div.message_content, div.content.allwith div.news-cont')
    if content_div:
        content_html = str(content_div)
    if not content_div or len(content_html) < 50:
        content_div = soup.select_one('div.zwnr, div.TRS_UEDITOR, div#zoom')
        if content_div:
            content_html = str(content_div)
    
    content_text = clean_content(content_html)
    
    return title, date, content_text

def clean_text(text):
    if not text:
        return ""
    text = re.sub(r'(&middot;|&nbsp;|\xa0|\u3000|\s)+', ' ', text)
    return text.strip()

def clean_content(html_content):
    """提取纯文本：按p分段，span间用空格（防TRS嵌套span分段过多）"""
    if not html_content:
        return ""
    soup = BeautifulSoup(html_content, 'html.parser')
    for tag in soup(['script', 'style', 'link']):
        tag.decompose()
    
    # 按<p>标签分段，每个<p>内合并所有span（空格分隔）
    paragraphs = []
    for p in soup.find_all('p'):
        # 去除p内的子标签（span/strong等），用空格连接文本
        text = ''.join(p.find_all(string=True, recursive=True))
        text = re.sub(r'\s+', ' ', text).strip()
        if text:
            paragraphs.append(text)
    
    if not paragraphs:
        # 回退：没有p标签时用全文
        text = soup.get_text(separator=' ')
        text = re.sub(r'\s+', ' ', text)
        return text.strip()
    
    return '\n\n'.join(paragraphs)

def push_to_db(items):
    if not items:
        return 0, 0
    new_count = 0
    skip_count = 0
    for item in items:
        title = clean_text(item.get("title", ""))
        date = item.get("date", "")
        url = item.get("url", "")
        content_text = item.get("content", "")
        if not title or not url:
            skip_count += 1
            continue
        pub_date = date
        dm = re.search(r'(\d{4})年(\d{1,2})月(\d{1,2})日', date)
        if dm:
            pub_date = f"{dm.group(1)}-{dm.group(2).zfill(2)}-{dm.group(3).zfill(2)}"
        
        sql = f"""INSERT OR IGNORE INTO gov_raw (page_url, source_url, title, publish_date, content, site_name, group_name, industry)
VALUES ('{url.replace("'","''")}', '{url.replace("'","''")}', '{title.replace("'","''")}',
'{pub_date}', '{content_text[:50000].replace("'","''")}',
'{SITE_NAME.replace("'","''")}', '{GROUP}', '政府公告');
"""
        r = subprocess.run(
            ["sqlite3", "-cmd", ".timeout 60000", DB_PATH],
            input=sql, capture_output=True, text=True, timeout=30
        )
        if r.returncode == 0 and "UNIQUE" not in r.stderr:
            new_count += 1
        else:
            skip_count += 1
    return new_count, skip_count

def sync_fts():
    sql = """
INSERT OR IGNORE INTO gov_search (rowid, title, site_name, summary)
SELECT r.rowid, r.title, r.site_name, substr(r.content, 1, 200)
FROM gov_raw r
LEFT JOIN gov_search s ON r.rowid = s.rowid
WHERE s.rowid IS NULL;
"""
    r = subprocess.run(
        ["sqlite3", "-cmd", ".timeout 60000", DB_PATH],
        input=sql, capture_output=True, text=True, timeout=60
    )
    return r.returncode

def main():
    all_items = []
    seen_urls = set()
    
    print(f"站点: {SITE_NAME}")
    print(f"群组: {GROUP} | 页数: {MAX_PAGES}")
    
    # 爬取主分类
    print(f"\n[1/2] 爬取主分类 (channelid={CHANNELID})...")
    html = fetch_list("3441", 1)
    items = parse_items(html)
    for it in items:
        if it["url"] not in seen_urls:
            seen_urls.add(it["url"])
            all_items.append(it)
    print(f"  主分类获取: {len(items)} 条")
    
    # 子分类
    print(f"\n[2/2] 爬取 {len(SUBCATEGORIES)} 个子分类...")
    for cid, cname in SUBCATEGORIES.items():
        html = fetch_list(cid, 1)
        items = parse_items(html)
        new_for_cat = 0
        for it in items:
            if it["url"] not in seen_urls:
                seen_urls.add(it["url"])
                all_items.append(it)
                new_for_cat += 1
        print(f"  {cname} (cid={cid}): +{new_for_cat} 条")
    
    print(f"\n总共获取: {len(all_items)} 条")
    
    if not all_items:
        print("没有数据需要处理")
        return
    
    # 获取详情
    print(f"\n正在获取详情页 ({len(all_items)} 条)...")
    success = 0
    for i, item in enumerate(all_items):
        html = fetch_detail(item["url"])
        if not html or len(html) < 500:
            continue
        title, date, content_text = extract_detail(html, item["url"])
        item["title"] = title or item["title"]
        item["date"] = date or item["date"]
        item["content"] = content_text
        success += 1
        if (i+1) % 10 == 0:
            print(f"  进度: {i+1}/{len(all_items)}")
    
    print(f"详情获取成功: {success}/{len(all_items)}")
    
    # 入库
    print(f"\n正在入库...")
    new_count, skip_count = push_to_db(all_items)
    print(f"新增: {new_count} | 跳过: {skip_count}")
    
    # FTS
    print(f"\n同步FTS...")
    sync_fts()
    
    r = subprocess.run(
        ["sqlite3", "-cmd", ".timeout 60000", DB_PATH],
        input="SELECT COUNT(*) FROM gov_search g JOIN gov_raw r ON g.rowid=r.rowid WHERE r.site_name='{}';".format(SITE_NAME.replace("'","''")),
        capture_output=True, text=True, timeout=15
    )
    fts_count = r.stdout.strip()
    print(f"FTS验证: {fts_count} 条")
    
    print(f"\n✅ 完成! {new_count} 新增, {skip_count} 跳过")

if __name__ == "__main__":
    main()
