#!/usr/bin/env python3
import re, sys, os, json, time, requests

DB_PATH = os.environ.get("DB_PATH", "/root/search.db")

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/webp,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}

SITE_NAME = "漳浦县人民政府-公示公告"
GROUP = "漳浦"
PER_PAGE = 20
MAX_PAGES = 5
BASE_URL = "http://www.zhangpu.gov.cn"
SITE_ID = "60423209134990000"
LIST_REFERER = BASE_URL + "/cms/html/zpxrmzf/gsgg/index.html"

def fetch_page(url, referer=None, encoding="utf-8"):
    headers = dict(HEADERS)
    if referer:
        headers["Referer"] = referer
    r = requests.get(url, headers=headers, timeout=30, allow_redirects=True)
    r.encoding = encoding
    return r.text

def resolve_url(rel_url):
    if rel_url.startswith("http"):
        return rel_url
    if rel_url.startswith("/"):
        return BASE_URL + rel_url
    return BASE_URL + "/" + rel_url

def extract_list_items(html):
    items = []
    # <LI><SPAN class="list-content"><A href="URL" target="_blank">TITLE</A></SPAN><SPAN class="list-time">DATE</SPAN></LI>
    pattern = r'<LI><SPAN class="list-content"><A href="([^"]+)"[^>]*>([^<]+)</A></SPAN><SPAN class="list-time">([^<]+)</SPAN></LI>'
    for m in re.finditer(pattern, html, re.DOTALL):
        rel_url = m.group(1).strip()
        title = m.group(2).strip()
        date_str = m.group(3).strip()
        full_url = resolve_url(rel_url)
        if title:
            items.append({"url": full_url, "title": title, "date": date_str})
    return items

def extract_content(html, source_url):
    # Title from meta
    title = ""
    m = re.search(r'ArticleTitle"\s*content="([^"]+)"|content="([^"]+)"\s*name="ArticleTitle"', html)
    if m:
        title = (m.group(1) or m.group(2)).strip()
    if not title:
        m = re.search(r"<[Tt][Ii][Tt][Ll][Ee]>([^<]+)", html)
        if m:
            title = m.group(1).strip()
            # Remove trailing " - 公示公告" etc.
            title = re.sub(r"\s*-\s*.*$", "", title)

    # Date from meta
    pub_date = ""
    m = re.search(r'PubDate"\s*content="([^"]+)"|content="([^"]+)"\s*name="PubDate"', html)
    if m:
        raw = (m.group(1) or m.group(2)).strip()
        # Handle various date formats like "2026—07—01 11:30"
        dm = re.search(r"(\d{4}[—\-/]\d{1,2}[—\-/]\d{1,2})", raw)
        if dm:
            pub_date = dm.group(1).replace("—", "-").replace("/", "-")

    # Content from <div id="content">
    content = ""
    attachments = []
    content_html = ""

    m = re.search(r'id="content"[^>]*>(.*?)</div>', html, re.DOTALL | re.I)
    if m:
        content_html = m.group(1)

    if content_html:
        # Extract attachments
        for am in re.finditer(r'<a[^>]*href=(["\x27])([^"\x27]*\.(?:pdf|doc|docx|xls|xlsx|rar|zip))\1[^>]*>(.*?)</a>', content_html, re.I | re.DOTALL):
            link_url = resolve_url(am.group(2))
            link_text = re.sub(r'<[^>]+>', "", am.group(3)).strip()
            attachments.append({"name": link_text or link_url.split("/")[-1], "url": link_url})

        # FIRST PASS: extract <table> tags, convert to Markdown, replace with placeholders
        # (tables may be nested inside <p> tags in this CMS, so we must extract them first)
        tables_md = {}
        def table_to_md(m):
            # 保留原始 HTML 表格（不转 md），占位符还原时插入原始表格
            idx = len(tables_md) + 1
            tables_md[idx] = m.group(0)
            return "\n__TBL_%d__\n" % idx

        processed_html = re.sub(r'<table[^>]*>.*?</table>', table_to_md, content_html, flags=re.DOTALL | re.I)

        # SECOND PASS: extract text parts from <p>, <div> (now table-free)
        parts = []
        for elem in re.finditer(r'<p[^>]*>(.*?)</p>|<div[^>]*>(.*?)</div>', processed_html, re.DOTALL | re.I):
            tag_start = elem.group(0)[:4].lower()
            inner = elem.group(1) or elem.group(2) or ""

            # Check if this element contains a table placeholder
            ph = re.search(r'__TBL_(\d+)__', inner)
            if ph:
                parts.append("__TBL_%s__" % ph.group(1))
                continue

            if tag_start.startswith("<div"):
                # Skip nav/toolbar divs
                div_text = re.sub(r'<[^>]+>', "", inner)
                div_text = re.sub(r'&[nN][bB][sS][pP];', " ", div_text)
                div_text = re.sub(r'\u3000', "", div_text)
                div_text = div_text.strip()
                if div_text and len(div_text) > 5:
                    sub_html = elem.group(0)
                    if "<br" in sub_html.lower():
                        br_parts = re.split(r'<br\s*/?\s*>', sub_html, flags=re.I)
                        for bp in br_parts:
                            bt = re.sub(r'<[^>]+>', "", bp).strip()
                            bt = re.sub(r'&[nN][bB][sS][pP];', " ", bt)
                            bt = re.sub(r'\u3000', "", bt)
                            if bt:
                                parts.append(bt)
                    else:
                        parts.append(div_text)
            else:
                # <p> tag
                p_text = re.sub(r'<[^>]+>', "", inner)
                p_text = re.sub(r'&[nN][bB][sS][pP];', " ", p_text)
                p_text = re.sub(r'\u3000', "", p_text)
                p_text = p_text.strip()
                if p_text:
                    parts.append(p_text)

        # Fallback: if no structured parts found
        if not parts:
            text = re.sub(r'<script[^>]*>.*?</script>', "", processed_html, flags=re.DOTALL | re.I)
            text = re.sub(r'<style[^>]*>.*?</style>', "", processed_html, flags=re.DOTALL | re.I)
            text = re.sub(r'<br\s*/?\s*>', "\n", text, flags=re.I)
            text = re.sub(r'<[^>]+>', "", text)
            text = re.sub(r'&[nN][bB][sS][pP];', " ", text)
            text = re.sub(r'\u3000', "", text)
            for line in text.split("\n"):
                line = line.strip()
                if line:
                    parts.append(line)

        # THIRD PASS: replace table placeholders with actual Markdown tables
        final_parts = []
        for p in parts:
            ph = re.search(r'__TBL_(\d+)__', p)
            if ph:
                final_parts.append(tables_md[int(ph.group(1))])
            else:
                final_parts.append(p)

        content = "\n\n".join(final_parts)

    return {"title": title, "content": content, "date": pub_date, "url": source_url,
            "attachments": attachments, "site_name": SITE_NAME, "group": GROUP}

def crawl(test_mode=False):
    import sqlite3
    all_items = []
    seen_urls = set()

    for page in range(1, MAX_PAGES + 1):
        if page == 1:
            url = "http://www.zhangpu.gov.cn/cms/html/zpxrmzf/gsgg/index.html"
        else:
            url = "http://www.zhangpu.gov.cn/cms/sitemanage/index.shtml?siteId=%s&page=%d" % (SITE_ID, page)

        html = fetch_page(url, referer=LIST_REFERER)
        items = extract_list_items(html)
        print("[Page %d] %s" % (page, url))
        print("  Found %d items" % len(items))
        for item in items:
            if item["url"] not in seen_urls:
                seen_urls.add(item["url"])
                all_items.append(item)
        if len(items) == 0:
            print("  No items found, stopping pagination")
            break

    print("\nTotal unique items: %d" % len(all_items))

    if test_mode:
        for item in all_items[:3]:
            print("\n=== Detail: %s ===" % item["title"][:50])
            html = fetch_page(item["url"], referer=LIST_REFERER)
            result = extract_content(html, item["url"])
            print("  Title: %s" % result["title"])
            print("  Date: %s" % result["date"])
            print("  Content (%d chars): %s" % (len(result["content"]), result["content"][:200]))
            print("  Attachments: %d" % len(result["attachments"]))
            for att in result["attachments"]:
                print("    - %s: %s" % (att["name"], att["url"][:80]))
        return

    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=30000")
    c = conn.cursor()
    inserted = 0
    existing = 0
    total = len(all_items)

    for idx, item in enumerate(all_items):
        c.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
        if c.fetchone():
            existing += 1
            continue
        html = fetch_page(item["url"], referer=LIST_REFERER)
        result = extract_content(html, item["url"])
        attachments_json = json.dumps(result["attachments"], ensure_ascii=False) if result["attachments"] else ""
        summary = result["title"] + " " + SITE_NAME
        if result["content"]:
            summary += " " + result["content"][:200]
        c.execute("INSERT INTO gov_raw (title, content, summary, publish_date, page_url, site_name, attachments, group_name, source_url) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
            (result["title"], result["content"], summary, result["date"] or item["date"], result["url"], result["site_name"], attachments_json, result["group"], result["url"]))
        inserted += 1
        if inserted % 5 == 0:
            conn.commit()
            print("  Progress: %d/%d inserted..." % (inserted, total))
    conn.commit()
    conn.close()
    print("\nDone. Inserted: %d, Existing: %d (total: %d)" % (inserted, existing, total))

if __name__ == "__main__":
    if "--test" in sys.argv:
        crawl(test_mode=True)
    else:
        crawl(test_mode=False)
