#!/usr/bin/env python3
"""
安庆市生态环境局 - 政府信息公开（主动公开）
https://sthjj.anqing.gov.cn/public/column/4018218?type=4
CMS: 龙讯Lonsun, AJAX分页104页(~1560条)
列表API: /public/column/4018218?type=4&action=list&isJson=true&pageIndex=N&pageSize=20
详情: div#zoom.j-fontContent 正文
"""

import sys, os, re, json, requests, time
from bs4 import BeautifulSoup

BASE_URL = "https://sthjj.anqing.gov.cn"
LIST_API = "https://sthjj.anqing.gov.cn/public/column/4018218"
SITE_NAME = "安庆市生态环境局-信息公开"
MAX_PAGES = 5  # 初始前5页

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def extract_table_tags(html_str):
    """从HTML字符串中提取所有<table>标签，用唯一标记替换"""
    markers = []
    def replace_table(m):
        idx = len(markers)
        key = f"___TBL_{idx}___"
        markers.append((key, m.group(0)))
        return key
    modified = re.sub(
        r"<table[^>]*>.*?</table>",
        replace_table,
        html_str,
        flags=re.DOTALL | re.IGNORECASE
    )
    return modified, markers


def fetch_list_page(page_index):
    """获取列表页数据"""
    params = {
        "type": "4",
        "action": "list",
        "isJson": "true",
        "siteId": "3902127",
        "organId": "4018218",
        "pageSize": "20",
        "pageIndex": str(page_index),
        "isDate": "true",
        "dateFormat": "yyyy-MM-dd",
        "length": "50",
        "labelName": "publicInfoList",
    }
    try:
        r = requests.get(LIST_API, params=params, headers=HEADERS, timeout=20)
        r.raise_for_status()
        # Response is JSON string encoding HTML
        html = r.json() if r.text.startswith('"') else r.text
    except Exception as e:
        print(f"  API ERROR: {e}", file=sys.stderr)
        return [], 0

    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.find_all("li", class_="clearfix"):
        a = li.find("a", class_="title", href=True)
        if not a:
            continue
        href = a["href"].strip()
        title = a.get("title", "") or a.get_text(strip=True)
        if not title or not href:
            continue
        full_url = href if href.startswith("http") else f"{BASE_URL}{href}"
        date_span = li.find("span", class_="date")
        date_str = date_span.get_text(strip=True) if date_span else ""
        items.append({"title": title.strip(), "url": full_url, "date": date_str})
    return items


def get_content(url):
    """获取详情页正文"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        r.raise_for_status()
    except Exception as e:
        print(f"  ERROR fetching detail: {e}", file=sys.stderr)
        return "", "", [], ""

    soup = BeautifulSoup(r.text, "html.parser")
    full_title = ""
    pub_date = ""

    # 标题
    title_tag = soup.find("title")
    if title_tag:
        t = title_tag.get_text(strip=True)
        for sep in ["_安庆市人民政府信息公开", "_安庆市人民政府", "_"]:
            idx = t.find(sep)
            if idx > 0:
                t = t[:idx]
                break
        full_title = t.strip()

    # 日期
    for m in re.finditer(r'PubDate[^>]+content="(\d{4}-\d{1,2}-\d{1,2})"', r.text):
        pub_date = m.group(1)
        break
    if not pub_date:
        dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", r.text[:3000])
        if dm:
            pub_date = dm.group(1)

    # 正文
    content_html = ""
    zoom = soup.find(id="zoom") or soup.select_one(".j-fontContent.newscontnet")

    attachments = []
    if zoom:
        # 替换附件链接
        for a_tag in zoom.find_all("a", href=True):
            h = a_tag["href"].strip()
            if re.search(r"\.(docx?|pdf|pptx?)($|\?)", h, re.IGNORECASE):
                fname = a_tag.get_text(strip=True) or h.split("/")[-1]
                attach_url = h if h.startswith("http") else f"{BASE_URL}{h}"
                attachments.append({"url": attach_url, "name": fname})
                md_link = f"[{fname}]({attach_url})"
                a_tag.replace_with(md_link)

        # 保存表格
        zoom_html = str(zoom)
        zoom_html, table_markers = extract_table_tags(zoom_html)

        body_soup = BeautifulSoup(zoom_html, "html.parser")
        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()
        for br in body_soup.find_all("br"):
            br.replace_with("\n")

        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()

        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, f"\n\n{tbl_html}\n\n")

    if not content_html or len(content_html.strip()) < 20:
        content_html = f'<p><a href="{url}">{full_title or "无正文"}</a></p>'

    return content_html, pub_date, attachments, full_title


def main():
    # 获取总页数
    items_page1 = fetch_list_page(1)
    total_pages = min(MAX_PAGES, 104)  # 前5页
    print(f"第1页: {len(items_page1)} 条, 总页: ~104, 爬前{total_pages}页", file=sys.stderr)

    all_items = []
    for pi in range(1, total_pages + 1):
        items = fetch_list_page(pi) if pi > 1 else items_page1
        if not items:
            break
        print(f"第{pi}页: {len(items)} 条", file=sys.stderr)

        for i, item in enumerate(items):
            print(f"  [{pi}.{i+1}] {item['title'][:40]}...", end=" ", file=sys.stderr)
            content, pub_date, attachments, full_title = get_content(item["url"])
            date = item["date"] or pub_date
            use_title = full_title or item["title"]
            all_items.append({
                "title": use_title[:500],
                "page_url": item["url"],
                "content": content,
                "publish_date": date[:20] if date else "",
                "site_name": SITE_NAME,
                "source_url": item["url"],
                "status": "synced",
                "attachments": json.dumps(attachments, ensure_ascii=False),
            })
            print(f"OK (body:{len(content)}B, attach:{len(attachments)})", file=sys.stderr)
            time.sleep(0.5)

    if not all_items:
        print("no data", file=sys.stderr)
        return

    # 写入search.db
    import sqlite3
    DB_PATH = "/root/search.db"
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=OFF")

    added = 0
    for item in all_items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?,?,?,?,?,?,?,?)",
                (item["title"][:500], item["page_url"][:1000], item["content"],
                 item["publish_date"][:20], item["site_name"], item["page_url"],
                 "synced", item["attachments"])
            )
            if db.total_changes > 0:
                added += 1
        except Exception as e:
            print(f"  DB ERROR: {e}", file=sys.stderr)
    db.commit()

    # 增量FTS
    db.execute("""
        INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary)
        SELECT r.rowid, r.title, r.site_name, substr(r.content, 1, 500)
        FROM gov_raw r
        WHERE r.site_name = ?
        AND r.content IS NOT NULL AND r.content != ''
        AND r.rowid NOT IN (SELECT rowid FROM gov_search)
    """, (SITE_NAME,))
    db.commit()
    db.close()

    print(f"DB: +{added} / {len(all_items)} items", file=sys.stderr)

    for item in all_items:
        item["attachments"] = json.loads(item["attachments"])
        print(json.dumps(item, ensure_ascii=False))

    # 日跑配置也写入JSONL供参考
    jsonl_path = f"/tmp/crawl_aqsthj_xxgk_{int(time.time())}.jsonl"
    with open(jsonl_path, "w", encoding="utf-8") as f:
        for item in all_items:
            item["attachments"] = json.dumps(item["attachments"], ensure_ascii=False)
            f.write(json.dumps(item, ensure_ascii=False) + "\n")
    print(f"JSONL backup: {jsonl_path}", file=sys.stderr)


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"time: {time.time()-t0:.1f}s", file=sys.stderr)
