#!/usr/bin/env python3
"""
安庆市生态环境局 - 调查征集（征集公告）
https://sthjj.anqing.gov.cn/hdjl/dczj/index.html
CMS: 龙讯Lonsun, 单页16条
列表: <a href="/content/article/{ID}" title="完整标题">标题</a>
日期: <span class="date">2025-09-29 至 2025-10-28</span>
详情: h1.ls-article-title + div.j-fontContent.newscontnet
附件: /group4/M00/... .doc/.docx/.pdf
"""

import sys, os, re, json, requests, time
from bs4 import BeautifulSoup

BASE_URL = "https://sthjj.anqing.gov.cn"
LIST_URL = "https://sthjj.anqing.gov.cn/hdjl/dczj/index.html"
SITE_NAME = "安庆市生态环境局-调查征集"

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

TBL_MARKER_PREFIX = "___TBL_"
TBL_MARKER_SUFFIX = "___"


def extract_table_tags(html_str):
    """从HTML字符串中提取所有<table>标签，用唯一标记替换"""
    markers = []
    def replace_table(m):
        idx = len(markers)
        key = f"{TBL_MARKER_PREFIX}{idx}{TBL_MARKER_SUFFIX}"
        markers.append((key, m.group(0)))
        return key
    modified = re.sub(
        r"<table[^>]*>.*?</table>",
        replace_table,
        html_str,
        flags=re.DOTALL | re.IGNORECASE
    )
    return modified, markers


def parse_list(html):
    """解析列表页"""
    items = []
    soup = BeautifulSoup(html, "html.parser")
    ul = soup.select_one("ul.collect-list")
    if not ul:
        print("WARN: collect-list ul not found", file=sys.stderr)
        return items
    for li in ul.find_all("li", recursive=False):
        a = li.find("a", href=True, class_="title2")
        if not a:
            continue
        href = a["href"].strip()
        title = a.get("title", "").strip() or a.get_text(strip=True)
        if not title:
            continue
        full_url = href if href.startswith("http") else f"{BASE_URL}{href}"
        date_span = li.find("span", class_="date")
        date_str = ""
        if date_span:
            dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", date_span.get_text())
            if dm:
                date_str = dm.group(1)
        items.append({"title": title, "url": full_url, "date": date_str})
    return items


def get_content(url):
    """获取详情页正文"""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
        r.raise_for_status()
    except Exception as e:
        print(f"  ERROR fetching detail: {e}", file=sys.stderr)
        return "", "", [], ""

    soup = BeautifulSoup(r.text, "html.parser")
    full_title = ""
    pub_date = ""

    # 标题
    h1 = soup.select_one("h1.ls-article-title")
    if h1:
        full_title = h1.get_text(strip=True)

    # 日期 - 从meta或页面取
    for m in re.finditer(r'PubDate[^>]+content="(\d{4}-\d{1,2}-\d{1,2})"', r.text):
        pub_date = m.group(1)
        break
    if not pub_date:
        dm = re.search(r"(\d{4}-\d{1,2}-\d{1,2})", r.text[:3000])
        if dm:
            pub_date = dm.group(1)

    # 正文 - 征集公告tab内容
    content_html = ""
    body_div = soup.select_one("div.j-fontContent.newscontnet")
    if not body_div:
        # Fallback
        body_div = soup.select_one("#aa_tabs1 .j-fontContent, .ls-article-info .j-fontContent")

    attachments = []
    if body_div:
        # 1) 提取附件链接
        for a_tag in body_div.find_all("a", href=True):
            h = a_tag["href"].strip()
            if re.search(r"\.(docx?|pdf|pptx?)($|\?)", h, re.IGNORECASE):
                fname = a_tag.get_text(strip=True) or h.split("/")[-1]
                attach_url = h if h.startswith("http") else f"{BASE_URL}{h}"
                attachments.append({"url": attach_url, "name": fname})
                # 替换a标签为Markdown链接
                md_link = f"[{fname}]({attach_url})"
                a_tag.replace_with(md_link)

        # 2) 保存表格，替换为标记
        body_html = str(body_div)
        body_html, table_markers = extract_table_tags(body_html)

        # 3) 重新解析
        body_soup = BeautifulSoup(body_html, "html.parser")

        # 4) 移除script/style
        for tag in body_soup.find_all(["script", "style"]):
            tag.decompose()

        # 5) unwrap内联标签
        for tag in body_soup.find_all(["span", "font", "b", "strong", "i", "em", "u"]):
            tag.unwrap()

        # 6) p标签前加分隔
        for p in body_soup.find_all("p"):
            p.insert(0, "\n\n")
            p.unwrap()
        # 7) div也加分隔
        for d in body_soup.find_all("div"):
            d.insert(0, "\n")
            d.unwrap()

        # 8) br转换行
        for br in body_soup.find_all("br"):
            br.replace_with("\n")

        # 9) 提取纯文本
        content_html = body_soup.get_text(separator="", strip=False)
        content_html = re.sub(r"\n{3,}", "\n\n", content_html).strip()

        # 10) 恢复表格HTML
        for key, tbl_html in table_markers:
            content_html = content_html.replace(key, f"\n\n{tbl_html}\n\n")

    if not content_html or len(content_html.strip()) < 20:
        content_html = f'<p><a href="{url}">{full_title or "无正文"}</a></p>'

    return content_html, pub_date, attachments, full_title


def main():
    r = requests.get(LIST_URL, headers=HEADERS, timeout=20)
    r.encoding = "utf-8"
    items = parse_list(r.text)
    print(f"列表: {len(items)} 条", file=sys.stderr)

    all_items = []
    for i, item in enumerate(items):
        print(f"  [{i+1}/{len(items)}] {item['title'][:40]}...", end=" ", file=sys.stderr)
        content, pub_date, attachments, full_title = get_content(item["url"])
        date = item["date"] or pub_date
        use_title = full_title or item["title"]
        all_items.append({
            "title": use_title[:500],
            "page_url": item["url"],
            "content": content,
            "publish_date": date[:20] if date else "",
            "site_name": SITE_NAME,
            "source_url": item["url"],
            "status": "synced",
            "attachments": json.dumps(attachments, ensure_ascii=False),
        })
        print(f"OK (body:{len(content)}B, attach:{len(attachments)})", file=sys.stderr)
        time.sleep(0.5)

    if not all_items:
        print("no data", file=sys.stderr)
        return

    # 直接写入 search.db（避免 import_jsonl.py FTS重建超时）
    import sqlite3
    DB_PATH = "/root/search.db"
    db = sqlite3.connect(DB_PATH, timeout=60)
    db.execute("PRAGMA journal_mode=WAL")
    db.execute("PRAGMA synchronous=OFF")

    added = 0
    for item in all_items:
        try:
            db.execute(
                "INSERT OR IGNORE INTO gov_raw (title, page_url, content, publish_date, site_name, source_url, status, attachments) VALUES (?,?,?,?,?,?,?,?)",
                (item["title"][:500], item["page_url"][:1000], item["content"],
                 item["publish_date"][:20], item["site_name"], item["page_url"],
                 "synced", item["attachments"])
            )
            if db.total_changes > 0:
                added += 1
        except Exception as e:
            print(f"  DB ERROR: {e}", file=sys.stderr)
    db.commit()

    # 增量FTS更新
    db.execute("""
        INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary)
        SELECT r.rowid, r.title, r.site_name, substr(r.content, 1, 500)
        FROM gov_raw r
        WHERE r.site_name = ?
        AND r.content IS NOT NULL AND r.content != ''
        AND r.rowid NOT IN (SELECT rowid FROM gov_search)
    """, (SITE_NAME,))
    db.commit()
    db.close()

    print(f"DB: +{added} / {len(all_items)} items", file=sys.stderr)

    for item in all_items:
        item["attachments"] = json.loads(item["attachments"])
        print(json.dumps(item, ensure_ascii=False))


if __name__ == "__main__":
    t0 = time.time()
    main()
    print(f"time: {time.time()-t0:.1f}s", file=sys.stderr)
