#!/usr/bin/env python3
"""泰兴市自然资源和规划局 — 建设项目批前公示

TRS CMS
列表: https://zrzy.jiangsu.gov.cn/tztx/gtzx/tzgg_10466/tzgg_10469/
分页: index_1.htm, index_2.htm ... (15条/页, 共26页)
详情: /202607/t20260715_2050484.htm
正文: div#fontzoom > div.TRS_Editor
"""

import re, sys, os, json, time
import urllib.request, urllib.error
from bs4 import BeautifulSoup
import sqlite3

BASE_URL = "https://zrzy.jiangsu.gov.cn/tztx/gtzx/tzgg_10466/tzgg_10469"
SITE_NAME = "泰兴市-建设项目批前公示"
SITE_DISPLAY = "泰兴市自然资源和规划局"
DB = os.environ.get("SEARCH_DB", "/root/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36"
}

_MAX_PG = None
for i, a in enumerate(sys.argv):
    if a == "--pages" and i + 1 < len(sys.argv):
        _MAX_PG = int(sys.argv[i + 1])
        break


def fetch(url):
    req = urllib.request.Request(url, headers=HEADERS)
    try:
        return urllib.request.urlopen(req, timeout=30).read().decode("utf-8", errors="replace")
    except Exception as e:
        print("[WARN] fetch failed: %s - %s" % (url, e), file=sys.stderr)
        return ""


def extract_list_items(html):
    """解析列表页，返回 (title, date, url) 列表"""
    items = []
    soup = BeautifulSoup(html, "html.parser")

    # 列表在第一个table内的tr行中
    # 每条: <tr><td>· <a href="...">title</a></td><td>date</td></tr>
    # 查找包含链接的td，其父tr的第二个td是日期
    for a in soup.find_all("a", href=True):
        href = a["href"].strip()
        # 跳过非详情链接
        if not href.endswith(".htm") or "index" in href or "javascript" in href:
            continue
        # 只取包含"t202"或年份路径的详情链接
        if not re.search(r'/20\d{4}/t20', href) and not re.search(r'20\d{6}', href):
            continue

        title = a.get_text(strip=True)
        if not title or len(title) < 5:
            continue

        # 去掉 "· " 前缀
        title = re.sub(r'^[·\s]+', '', title).strip()

        # 找父td中隔壁的日期
        date_str = ""
        parent_td = a.find_parent("td")
        if parent_td:
            tr = parent_td.find_parent("tr")
            if tr:
                all_tds = tr.find_all("td")
                for td in all_tds:
                    txt = td.get_text(strip=True)
                    if re.match(r'\d{4}-\d{2}-\d{2}', txt):
                        date_str = txt
                        break

        # 补全URL
        if href.startswith("/"):
            full_url = "https://zrzy.jiangsu.gov.cn" + href
        elif not href.startswith("http"):
            full_url = BASE_URL.rstrip("/") + "/" + href.lstrip("./")
        else:
            full_url = href

        items.append((title.strip(), date_str, full_url))

    return items


def get_pagination_count(html):
    # JS: createPageHTML(26, 1, "index", "htm");
    m = re.search(r"createPageHTML\((\d+)", html)
    """解析总页数: '共26页'"""
    m = re.search(r'createPageHTML\((\d+)', html)
    if m:
        return int(m.group(1))
    return None


def parse_detail(html, url):
    """解析详情页"""
    soup = BeautifulSoup(html, "html.parser")

    # === Title ===
    title = ""
    # 标题在td font-size:23px bold
    title_td = soup.find("td", style=re.compile(r'font-size:\s*23'))
    if title_td:
        title = title_td.get_text(strip=True)
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            raw = title_tag.get_text(strip=True)
            title = re.sub(r'[_\-].*$', '', raw).strip()
    if not title:
        title = ""

    # === Date ===
    date_text = ""
    date_td = soup.find("td", class_="red_h")
    if date_td:
        txt = date_td.get_text(strip=True)
        m = re.search(r'(\d{4}-\d{2}-\d{2})', txt)
        if m:
            date_text = m.group(1)

    # === Content ===
    content_parts = []

    # 正文在 div#fontzoom > div.TRS_Editor
    fontzoom = soup.find("div", id="fontzoom")
    if not fontzoom:
        fontzoom = soup.find("div", id=re.compile(r'fontzoom|zoom|content', re.I))

    editor = None
    if fontzoom:
        editor = fontzoom.find("div", class_="TRS_Editor")
    if not editor:
        if fontzoom:
            editor = fontzoom
        else:
            editor = soup.find("div", class_="TRS_Editor")

    if editor:
        for el in editor.find_all(["p", "div", "table", "img"], recursive=True):
            # 跳过附件区域的div（附件会在最后统一处理）
            if el.find_parent("div") and el.find_parent("div") != editor:
                # 只在直接子元素层级处理
                pass

            if el.name == "img":
                src = el.get("src", "")
                alt = el.get("alt", "")
                if src:
                    if src.startswith("/"):
                        full_src = "https://zrzy.jiangsu.gov.cn" + src
                    elif src.startswith("./"):
                        full_src = url[:url.rfind("/")] + "/" + src[2:]
                    elif not src.startswith("http"):
                        if "/" in url[:url.rfind("/")]:
                            full_src = url[:url.rfind("/")] + "/" + src
                        else:
                            full_src = src
                    else:
                        full_src = src
                    content_parts.append("![%s](%s)" % (alt, full_src))
                continue

            if el.name == "table":
                rows = []
                for tr in el.find_all("tr"):
                    cells = [td.get_text(strip=True) for td in tr.find_all(["td", "th"])]
                    if any(cells):
                        rows.append(" | ".join(cells))
                if rows:
                    content_parts.append("\n".join(rows))
                continue

            # p 或 div
            txt = el.get_text(strip=True)
            if txt and len(txt) > 2:
                # 过滤无效文本
                if txt.startswith("附件：") or txt.startswith("附件"):
                    # 附件标题行，跳过（后面统一处理附件）
                    continue
                if re.match(r'^[\[（(]?\s*(大|中|小)\s*[\]）)]?\s*$', txt):
                    continue
                if "打印" in txt and "关闭" in txt:
                    continue
                if "字号" in txt and "[" in txt:
                    continue
                content_parts.append(txt)

    content = "\n\n".join(content_parts)

    # === Attachments ===
    attachments = []
    # 在TRS_Editor内的链接
    if editor:
        for a in editor.find_all("a", href=True):
            h = a["href"].lower()
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ppt|pptx)$', h):
                fname = a.get_text(strip=True) or os.path.basename(a["href"])
                # 补全URL
                full_url_path = ""
                if h.startswith("/"):
                    full_url_path = "https://zrzy.jiangsu.gov.cn" + a["href"]
                elif h.startswith("./"):
                    full_url_path = url[:url.rfind("/")] + "/" + h[2:]
                elif not h.startswith("http"):
                    base = url[:url.rfind("/")] if "/" in url else url
                    full_url_path = base + "/" + h
                else:
                    full_url_path = a["href"]
                attachments.append("[%s](%s)" % (fname, full_url_path))

    # 页面底部的附件区域（在fontzoom之外也可能有）
    footer_attach = soup.find(string=re.compile(r'附件[：:]'))
    if footer_attach:
        parent = footer_attach.find_parent("td") or footer_attach.find_parent("div")
        if parent:
            for a in parent.find_all("a", href=True):
                h = a["href"].lower()
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar|ppt|pptx)$', h):
                    fname = a.get_text(strip=True) or os.path.basename(a["href"])
                    if h.startswith("/"):
                        full_url_path = "https://zrzy.jiangsu.gov.cn" + a["href"]
                    elif h.startswith("./"):
                        full_url_path = url[:url.rfind("/")] + "/" + h[2:]
                    elif not h.startswith("http"):
                        base = url[:url.rfind("/")] if "/" in url else url
                        full_url_path = base + "/" + h
                    else:
                        full_url_path = a["href"]
                    attach_text = "[%s](%s)" % (fname, full_url_path)
                    if attach_text not in attachments:
                        attachments.append(attach_text)

    # 去重
    seen_att = set()
    unique_attachments = []
    for att in attachments:
        if att not in seen_att:
            seen_att.add(att)
            unique_attachments.append(att)

    if unique_attachments:
        attach_section = "\n\n---\n**附件：**\n" + "\n".join(unique_attachments)
        if not content:
            content = attach_section.lstrip()
        else:
            content += attach_section

    return title, date_text, content, "|".join(unique_attachments)


def store_item(title, date_str, content, page_url, attachments):
    try:
        conn = sqlite3.connect(DB, timeout=10)
        c = conn.cursor()
        summary = (content.strip()[:200] if content.strip() else "")
        date_rank = 0
        if date_str:
            try:
                from datetime import datetime
                date_rank = int(datetime.strptime(date_str[:10], "%Y-%m-%d").timestamp())
            except:
                pass
        c.execute("""INSERT OR IGNORE INTO gov_raw
                     (title, publish_date, content, page_url, source_url, site_name, summary, date_rank, attachments)
                     VALUES (?,?,?,?,?,?,?,?,?)""",
                  (title.strip(), date_str, content.strip(), page_url.strip(), page_url.strip(),
                   SITE_NAME, summary, date_rank, attachments))
        a = c.rowcount
        conn.commit()
        conn.close()
        return a
    except Exception as e:
        print("[ERROR] 写入失败: %s" % e, file=sys.stderr)
        return 0


def main():
    print("[START] %s" % SITE_NAME, file=sys.stderr)

    # 获取第1页，解析总页数
    first_html = fetch(BASE_URL + "/")
    if not first_html:
        print("[FATAL] 首页获取失败", file=sys.stderr)
        return

    total_pages = get_pagination_count(first_html)
    if total_pages:
        print("[INFO] 共 %d 页" % total_pages, file=sys.stderr)
    else:
        print("[WARN] 未能解析总页数，仅爬第1页", file=sys.stderr)
        total_pages = 1

    max_pg = total_pages
    if _MAX_PG and _MAX_PG < max_pg:
        max_pg = _MAX_PG
        print("[INFO] 限制为前 %d 页" % max_pg, file=sys.stderr)

    total_new = 0
    total_skip = 0
    total_err = 0

    for pg in range(1, max_pg + 1):
        if pg == 1:
            page_url = BASE_URL + "/"
        else:
            page_url = BASE_URL + "/index_%d.htm" % pg

        print("[PAGE] %d/%d: %s" % (pg, max_pg, page_url), file=sys.stderr)

        if pg == 1:
            html = first_html
        else:
            html = fetch(page_url)
            if not html:
                print("[ERR] 第%d页获取失败" % pg, file=sys.stderr)
                continue

        items = extract_list_items(html)
        print("[INFO] 第%d页 %d 条" % (pg, len(items)), file=sys.stderr)

        for title, date_str, detail_url in items:
            # 详情页内容
            detail_html = fetch(detail_url)
            if not detail_html:
                print("  [ERR] 详情页获取失败: %s" % detail_url, file=sys.stderr)
                total_err += 1
                continue

            d_title, d_date, d_content, attachments = parse_detail(detail_html, detail_url)
            # 优先使用列表页标题（更完整）
            final_title = d_title or title
            final_date = d_date or date_str

            if final_title and d_content:
                st = store_item(final_title, final_date, d_content, detail_url, attachments)
                if st:
                    total_new += 1
                    print("  [OK] %s %s" % (final_date, final_title[:50]), file=sys.stderr)
                else:
                    total_skip += 1
                    print("  [DUP] %s %s" % (final_date, final_title[:50]), file=sys.stderr)
            else:
                total_err += 1
                print("  [ERR] 解析失败: %s: title=%s, content=%s" %
                      (detail_url, bool(d_title), bool(d_content)), file=sys.stderr)

            time.sleep(0.3)

        time.sleep(0.5)

    print("[DONE] 新增: %d, 跳过: %d, 失败: %d" % (total_new, total_skip, total_err), file=sys.stderr)
    print("OK: new=%d skip=%d err=%d" % (total_new, total_skip, total_err))


if __name__ == "__main__":
    main()
