#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
第五师双河市-环评公示 爬虫 (CMS AJAX)
站点: www.xjshs.gov.cn
栏目: 环评公示 (catid=177, category=35)
API: POST api-ajax_list-{page}.html
"""

import sys, os, re, json, sqlite3, math, urllib.request, urllib.error, urllib.parse, warnings
warnings.filterwarnings('ignore')

_HERE = os.path.dirname(os.path.abspath(__file__))
sys.path.insert(0, _HERE)
sys.path.insert(0, os.path.join(_HERE, '..', 'crawler'))

from crawler_lib import fetch_page

SITE_NAME = "第五师双河市-环评公示"
DOMAIN = "www.xjshs.gov.cn"
BASE_URL = "https://www.xjshs.gov.cn"
SEARCH_DB = os.getenv("SEARCH_DB", "/root/search.db")

PAGE_SIZE = 20

# ─── DB ───
def insert_db(items):
    if not items:
        return
    db = sqlite3.connect(SEARCH_DB, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA busy_timeout=8000")
    db.execute("PRAGMA synchronous=NORMAL")
    db.execute("""CREATE TABLE IF NOT EXISTS gov_raw (
        id INTEGER PRIMARY KEY AUTOINCREMENT,
        site_name TEXT, source_url TEXT UNIQUE, page_url TEXT,
        title TEXT, publish_date TEXT, summary TEXT,
        content TEXT, status TEXT, category TEXT, tags TEXT,
        attachments TEXT
    )""")
    db.execute("""CREATE VIRTUAL TABLE IF NOT EXISTS gov_search USING fts5(
        title, site_name, summary, content='gov_raw', content_rowid='id'
    )""")
    db.commit()
    ok = skip = 0
    for item in items:
        try:
            title = (item.get("title") or "")[:500]
            summary = (item.get("summary") or title)[:500]
            content = item.get("content") or ""
            pub_date = (item.get("publish_date") or "")[:10]
            source_url = item.get("source_url") or ""
            page_url = item.get("url") or source_url
            attachments = item.get("attachments") or ""
            db.execute("""INSERT OR IGNORE INTO gov_raw
                (site_name, source_url, page_url, title, publish_date,
                 summary, content, status, attachments)
                VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
                (SITE_NAME, source_url, page_url, title, pub_date,
                 summary, content, "active", attachments))
            if db.total_changes > 0:
                ok += 1
            else:
                skip += 1
        except:
            skip += 1
    db.commit()
    try:
        db.execute("""INSERT OR REPLACE INTO gov_search(rowid, title, site_name, summary)
            SELECT r.id, r.title, r.site_name, r.summary
            FROM gov_raw r
            WHERE r.site_name=? AND r.id NOT IN (SELECT rowid FROM gov_search)""",
            (SITE_NAME,))
        db.commit()
    except Exception as e:
        print(f"  [FTS ERROR] {e}")
    db.close()
    print(f"  [DB] 新增: {ok}, 跳过: {skip}")
    return ok

def clean_content(html):
    if not html:
        return ""
    html = re.sub(r'<style[^>]*>.*?</style>', '', html, flags=re.DOTALL | re.IGNORECASE)
    html = re.sub(r'<script[^>]*>.*?</script>', '', html, flags=re.DOTALL | re.IGNORECASE)
    html = re.sub(r'<!--.*?-->', '', html, flags=re.DOTALL)
    return html.strip()

def extract_text(html):
    if not html:
        return ""
    text = html
    text = re.sub(r'<br\s*/?>', '\n', text, flags=re.IGNORECASE)
    # </p> → \n\n 确保段落间有空行
    text = re.sub(r'</p>', '\n\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</div>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</tr>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</th>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'</td>', '\n', text, flags=re.IGNORECASE)
    text = re.sub(r'<(?!/?(?:table|tr|td|th|thead|tbody|tfoot|img))[^>]*>', '', text)
    text = re.sub(r'&nbsp;', ' ', text)
    text = re.sub(r'&lt;', '<', text)
    text = re.sub(r'&gt;', '>', text)
    text = re.sub(r'&amp;', '&', text)
    # 连续 3+ 换行压缩为 \n\n（段落间空行）
    text = re.sub(r'\n{3,}', '\n\n', text)
    text = re.sub(r'[ \t]+', ' ', text)
    # 仅清理中文间水平空白（不碰段落换行）
    text = re.sub(r'(?<=[\u4e00-\u9fff])[ \t]+(?=[\u4e00-\u9fff])', '', text)
    return text.strip()

def extract_attachments(html, base_url=BASE_URL):
    if not html:
        return ""
    ext_pattern = r'\.(doc|docx|pdf|xls|xlsx|rar|zip)(\?[^\s"\'<>]*)?'
    links = re.findall(
        r'<a[^>]*href=["\']([^"\']*?(?:' + ext_pattern + r'))["\'][^>]*>([^<]*)</a>',
        html, re.IGNORECASE)
    if not links:
        return ""
    parts = []
    for href, _, _, text in links:
        full_url = href if href.startswith('http') else (base_url.rstrip('/') + '/' + href.lstrip('/'))
        parts.append(f"[附件: {text.strip()}]({full_url})")
    return "\n".join(parts)

# ─── API 请求 ───
def fetch_list(page=1):
    url = f"{BASE_URL}/api-ajax_list-{page}.html"
    post_data = (
        "ajax_type[]=18_xxgk&ajax_type[]=177&ajax_type[]=18&ajax_type[]=xxgk"
        "&ajax_type[]=Y-m-d&ajax_type[]=35&ajax_type[]=20"
        "&ajax_type[7][]=inputtime+DESC&ajax_type[7][]=displayorder+DESC"
        "&ajax_type[7][]=is_top+DESC&ajax_type[]=null&is_ds=1"
    )
    try:
        req = urllib.request.Request(url, data=post_data.encode())
        req.add_header("User-Agent", "Mozilla/5.0")
        req.add_header("Content-Type", "application/x-www-form-urlencoded")
        resp = urllib.request.urlopen(req, timeout=30)
        data = json.loads(resp.read())
    except Exception as e:
        print(f"  ❌ API请求失败(page={page}): {e}")
        return None, 0
    return data.get("data", []), data.get("total", 0)

# ─── 详情页 ───
def fetch_detail(url):
    """获取详情页内容，处理 302 重定向"""
    if not url or "xjshs.gov.cn" not in url:
        return None
    html = fetch_page(url, timeout=30)
    if not html:
        return None

    # 完整标题
    mt = re.search(r'<meta\s+name=["\']ArticleTitle["\'][^>]*content=["\']([^"\']+)["\']', html, re.IGNORECASE)
    detail_title = None
    if mt:
        detail_title = mt.group(1).strip()
    else:
        tt = re.search(r'<title>([^<]+)</title>', html, re.IGNORECASE)
        if tt:
            t = tt.group(1).strip()
            # 去掉 "第五师双河市人民政府-" 前缀
            t = re.sub(r'^第五师双河市人民政府\s*[-–—]\s*', '', t)
            detail_title = t

    # 提取 xxgk_content
    idx = html.find('class="xxgk_content"')
    if idx < 0:
        idx = html.find("class='xxgk_content'")
    if idx < 0:
        return None

    tag_end = html.find('>', idx)
    if tag_end < 0:
        return None
    rest = html[tag_end+1:]

    depth = 1
    pos = 0
    while depth > 0 and pos < len(rest):
        if rest[pos:pos+4] == '<div' and not rest[pos+4:pos+5] == '>':
            depth += 1
            pos += 4
        elif rest[pos:pos+6] == '</div>':
            depth -= 1
            pos += 6
        else:
            pos += 1
    raw_content = rest[:pos-6]

    cleaned = clean_content(raw_content)
    attachments = extract_attachments(raw_content)
    text_content = extract_text(cleaned)

    if attachments:
        text_content = text_content.rstrip() + "\n\n--- 附件 ---\n" + attachments

    return {
        "title": detail_title,
        "content": text_content,
        "attachments": attachments,
    }

def run(max_pages=None):
    print(f"\n{'='*60}")
    print(f"🚀 {SITE_NAME}")
    print(f"{'='*60}")

    # 第1页 + 总条数
    print("\n📥 获取列表...")
    articles, total = fetch_list(1)
    if articles is None:
        return
    total_pages = math.ceil(total / PAGE_SIZE)
    print(f"   总条数: {total}, 每页: {PAGE_SIZE}, 总页数: {total_pages}")

    if max_pages and max_pages > 0 and max_pages < total_pages:
        total_pages = max_pages
        print(f"   --pages 限制: 取前 {max_pages} 页")

    # 所有页
    all_items = list(articles) if articles else []
    for p in range(2, total_pages + 1):
        arts, _ = fetch_list(p)
        if arts:
            all_items.extend(arts)
            print(f"   第 {p}/{total_pages} 页: {len(arts)} 条")
        else:
            print(f"   第 {p}/{total_pages} 页: 空")

    print(f"   共获取: {len(all_items)} 条")

    # 逐条详情
    success = fail = 0
    db_items = []

    for i, item in enumerate(all_items, 1):
        url = item.get("url", "")
        title = item.get("title", "")
        date_str = item.get("inputtime", "")

        print(f"\n[{i}/{len(all_items)}] {title[:60]}...")

        detail = fetch_detail(url)
        if detail is None:
            fail += 1
            continue

        db_title = detail.get("title") or title
        content_text = detail["content"]

        db_items.append({
            "title": db_title,
            "source_url": url,
            "url": url,
            "publish_date": date_str[:10],
            "summary": db_title[:200],
            "content": content_text,
            "attachments": detail.get("attachments", ""),
        })
        success += 1
        print(f"   ✅ 正文: {len(content_text)} 字, 日期: {date_str[:10]}")

    print(f"\n{'='*60}")
    print(f"📦 写入数据库... 成功: {success}, 失败: {fail}, 跳过: {len(all_items)-success-fail}")
    if db_items:
        insert_db(db_items)
    print(f"{'='*60}")
    print(f"✅ 完成! 共处理 {len(all_items)} 条，成功 {success}，失败 {fail}")

if __name__ == "__main__":
    max_p = None
    args = sys.argv[1:]
    for i, arg in enumerate(args):
        if arg == "--pages" and i + 1 < len(args):
            try:
                max_p = int(args[i + 1])
            except:
                pass
    if max_p is None:
        for arg in args:
            try:
                max_p = int(arg.lstrip('-'))
                break
            except:
                pass
    run(max_pages=max_p)
