#!/usr/bin/env python3
"""
安庆市生态环境局 - 行政许可
https://www.anqing.gov.cn/public/column/4018218?type=4&catId=6999276&action=list
CMS: 龙讯Lonsun, JSON API分页
仅保留 ColumnName="行政许可" 的条目
"""
import sys, os, re, json, requests, time
from bs4 import BeautifulSoup
from urllib.parse import urljoin

BASE_URL = "https://sthjj.anqing.gov.cn"
LIST_API = "https://sthjj.anqing.gov.cn/public/column/4018218"
SITE_NAME = "安庆市生态环境局-行政许可"
MAX_PAGES = 5
TARGET_COLUMN = "行政许可"

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}


def extract_table_tags(html_str):
    markers = []
    def replace_table(m):
        idx = len(markers)
        key = f"___TBL_{idx}___"
        markers.append((key, m.group(0)))
        return key
    modified = re.sub(
        r"<table[^>]*>.*?</table>",
        replace_table,
        html_str,
        flags=re.DOTALL | re.IGNORECASE
    )
    return modified, markers


def fetch_list_page(page_index):
    """获取列表页数据"""
    params = {
        "type": "4",
        "action": "list",
        "isJson": "true",
        "siteId": "3902127",
        "organId": "4018218",
        "pageSize": "20",
        "pageIndex": str(page_index),
        "isDate": "true",
        "dateFormat": "yyyy-MM-dd",
        "length": "50",
        "labelName": "publicInfoList",
    }
    try:
        r = requests.get(LIST_API, params=params, headers=HEADERS, timeout=20)
        r.raise_for_status()
        html = r.json() if r.text.startswith('"') else r.text
    except Exception as e:
        print(f"  API ERROR: {e}", file=sys.stderr)
        return [], 0

    items = []
    soup = BeautifulSoup(html, "html.parser")
    for li in soup.find_all("li", class_="clearfix"):
        a = li.find("a", class_="title", href=True)
        if not a:
            continue
        href = a["href"].strip()
        title = a.get("title", "") or a.get_text(strip=True)
        if not title or not href:
            continue
        full_url = href if href.startswith("http") else f"{BASE_URL}{href}"
        date_span = li.find("span", class_="date")
        date_str = date_span.get_text(strip=True) if date_span else ""
        items.append({"title": title.strip(), "url": full_url, "date": date_str})

    # 尝试取总页数
    total_pages = 0
    m = re.search(r'total[Pp]age["\']?\s*[:=]\s*["\']?(\d+)', html)
    if m:
        total_pages = int(m.group(1))
    return items, total_pages


def get_content(url):
    """获取详情页正文，返回(content, pub_date, attachments, full_title, is_target)"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        r.raise_for_status()
    except Exception as e:
        print(f"  ERROR fetching detail: {e}", file=sys.stderr)
        return "", "", [], "", False

    soup = BeautifulSoup(r.text, "html.parser")

    # 检查是否目标栏目
    meta_col = soup.find("meta", attrs={"name": "ColumnName"})
    if meta_col and meta_col.get("content"):
        col_name = meta_col["content"].strip()
    else:
        col_name = ""
    if col_name != TARGET_COLUMN:
        return "", "", [], "", False

    full_title = ""
    pub_date = ""

    # 标题：meta ArticleTitle
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        full_title = meta_title["content"].strip()
    else:
        title_tag = soup.find("title")
        if title_tag:
            t = title_tag.get_text(strip=True)
            for sep in ["_安庆市人民政府信息公开", "_安庆市人民政府", "_"]:
                idx = t.find(sep)
                if idx > 0:
                    t = t[:idx]
                    break
            full_title = t.strip()

    # 日期：meta PubDate
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]
    if not pub_date:
        dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", r.text[:3000])
        if dm:
            pub_date = dm.group(1)

    # 正文
    content_html = ""
    zoom = soup.find(id="zoom") or soup.select_one(".j-fontContent.newscontnet")
    attachments = []

    if zoom:
        # 替换附件链接
        for a_tag in zoom.find_all("a", href=True):
            h = a_tag["href"].strip()
            if re.search(r"\.(docx?|pdf|pptx?)($|\?)", h, re.IGNORECASE):
                fname = a_tag.get_text(strip=True) or h.split("/")[-1]
                attach_url = h if h.startswith("http") else f"{BASE_URL}{h}"
                attachments.append({"url": attach_url, "name": fname})
                md_link = f"[{fname}]({attach_url})"
                a_tag.replace_with(md_link)

        # 保存表格
        zoom_html = str(zoom)
        zoom_html, table_markers = extract_table_tags(zoom_html)

        body_soup = BeautifulSoup(zoom_html, "html.parser")
        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()
        for br in body_soup.find_all("br"):
            br.replace_with("\n")

        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()

        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, f"\n\n{tbl_html}\n\n")

    if not content_html or len(content_html.strip()) < 20:
        content_html = f'<p><a href="{url}">{full_title or "无正文"}</a></p>'
    elif attachments:
        content_html += "\n\n**附件：**"
        for att in attachments:
            content_html += f"\n[{att['name']}]({att['url']})"

    return content_html, pub_date, attachments, full_title, True


def import_to_db(record):
    """写入search.db + FTS"""
    try:
        import sqlite3
        DB_PATH = "/root/search.db"
        db = sqlite3.connect(DB_PATH, timeout=60)
        db.execute("PRAGMA journal_mode=WAL")
        db.execute("PRAGMA synchronous=OFF")

        title = (record.get("title") or "")[:500]
        page_url = (record.get("page_url") or "")[:1000]
        content = record.get("content") or ""
        publish_date = (record.get("publish_date") or "")[:20]
        site_name = (record.get("site_name") or "unknown")[:100]
        attachments_str = record.get("attachments") or "[]"

        db.execute(
            "INSERT OR REPLACE INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments, script_name) VALUES (?,?,?,?,?,?,?,?,?)",
            (title, page_url, content, publish_date, site_name, page_url, "synced", attachments_str, "crawl_aqsthj_xzxk.py")
        )
        # FTS同步
        db.execute("""
            INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary)
            SELECT r.rowid, r.title, r.site_name, substr(r.content, 1, 500)
            FROM gov_raw r
            WHERE r.page_url = ?
        """, (page_url,))
        db.commit()
        db.close()
        return True
    except Exception as e:
        print(f"  DB ERROR: {e}", file=sys.stderr)
        return False


def main():
    items, total_pages = fetch_list_page(1)
    if total_pages == 0:
        total_pages = 104  # fallback
    total_pages = min(MAX_PAGES, total_pages)
    print(f"第1页: {len(items)} 条, 总页: ~{total_pages}", file=sys.stderr)

    all_items = []
    for pi in range(1, total_pages + 1):
        page_items, _ = fetch_list_page(pi) if pi > 1 else (items, total_pages)
        if not page_items:
            break
        print(f"第{pi}页: {len(page_items)} 条", file=sys.stderr)

        for i, item in enumerate(page_items):
            print(f"  [{pi}.{i+1}] {item['title'][:40]}...", end=" ", file=sys.stderr)
            content, pub_date, attachs, full_title, is_target = get_content(item["url"])
            if not is_target:
                print(f"SKIP (非{TARGET_COLUMN})", file=sys.stderr)
                continue

            date = item["date"] or pub_date
            use_title = full_title or item["title"]
            record = {
                "title": use_title[:500],
                "page_url": item["url"],
                "content": content,
                "publish_date": date[:20] if date else "",
                "site_name": SITE_NAME,
                "attachments": json.dumps(attachs, ensure_ascii=False),
            }
            ok = import_to_db(record)
            if ok:
                all_items.append(record)
            print(f"{'OK' if ok else 'FAIL'} (body:{len(content)}B)", file=sys.stderr)
            time.sleep(0.5)

    print(f"\n完成: 共入库 {len(all_items)} 条行政许可数据", file=sys.stderr)


if __name__ == "__main__":
    main()
