#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""台前县人民政府 - 政府公告爬虫"""
import sys, re, os, json, argparse
sys.path.insert(0, "/root")
sys.path.insert(0, "/root/gov_crawler")
from urllib.parse import urljoin
from datetime import datetime
import urllib.request, ssl

SITE_NAME = "台前县政府公告"
BASE_URL = "http://www.taiqian.gov.cn"
LIST_URL = "http://www.taiqian.gov.cn/channel/list/19256.html"

ssl_ctx = ssl.create_default_context()
ssl_ctx.check_hostname = False
ssl_ctx.verify_mode = ssl.CERT_NONE

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36",
    "Accept": "text/html,application/xhtml+xml",
}


def fetch_page(page_url):
    req = urllib.request.Request(page_url, headers=HEADERS)
    resp = urllib.request.urlopen(req, timeout=30, context=ssl_ctx)
    return resp.read().decode("utf-8")


def parse_list_page(html):
    from bs4 import BeautifulSoup
    soup = BeautifulSoup(html, "html.parser")
    items = []

    ul = soup.find("ul", class_="news-list")
    if not ul:
        return items

    for li in ul.find_all("li"):
        a = li.find("a", href=True)
        if not a:
            continue
        href = a["href"]
        text = a.get_text(strip=True)
        if not text or len(text) < 5:
            continue

        if href.startswith("/"):
            href = BASE_URL + href
        elif not href.startswith("http"):
            href = BASE_URL + "/" + href.lstrip("/")

        # Title has date prefix like "2026-07-08《台前县..." - strip date
        title_clean = re.sub(r"^\d{4}-\d{2}-\d{2}", "", text).strip()

        # Date from title prefix
        date_str = ""
        m = re.match(r"(\d{4}-\d{2}-\d{2})", text)
        if m:
            date_str = m.group(1)

        items.append({
            "url": href,
            "title": title_clean,
            "date": date_str,
        })
    return items


def fetch_detail(url):
    """抓取详情页，带3次重试"""
    import time
    for attempt in range(3):
        try:
            req = urllib.request.Request(url, headers=HEADERS)
            resp = urllib.request.urlopen(req, timeout=30, context=ssl_ctx)
            html = resp.read().decode("utf-8", errors="replace")
            # Check for error page
            if "Internal Server Error" in html[:500] or len(html) < 1000:
                if attempt < 2:
                    time.sleep(2 * (attempt + 1))
                    continue
                return "", "", [], ""
            break
        except Exception:
            if attempt < 2:
                time.sleep(2 * (attempt + 1))
                continue
            return "", "", [], ""

    from bs4 import BeautifulSoup
    import html as html_lib
    # ⚠️ source HTML defect: Word export replaced attribute separators with "+"
    #    e.g. <p+class="MsoNormal"+style="..."> → fix to <p class="MsoNormal" style="...">
    html = re.sub(r'<(/?(?:p|span|div|td|tr|table|h\d|a|img|li|ul|ol|br|font|strong|b|i|u))\s*\+', r'<\1 ', html)
    html = re.sub(r'\+([a-zA-Z-]+=)', r' \1', html)
    soup = BeautifulSoup(html, "html.parser")

    # Title: prefer h1.content-title (完整标题), fallback <title>
    title = ""
    h1 = soup.find("h1", class_="content-title")
    if h1:
        title = h1.get_text(" ", strip=True)
    if not title:
        m = re.search(r"<title>(.*?)</title>", html)
        if m:
            t = m.group(1).replace("台前县政府", "").replace("台前人民政府", "").strip()
            if t and t != "台前":
                title = t

    # Date
    date_str = ""
    for pat in [
        r"发布时间[：:]\s*(\d{4}-\d{2}-\d{2})",
        r"(\d{4}-\d{2}-\d{2}\s+\d{2}:\d{2})",
        r"(\d{4}-\d{2}-\d{2})",
    ]:
        m = re.search(pat, html)
        if m:
            date_str = m.group(1).replace("/", "-").replace(".", "-")[:10]
            break

    # Content - use div.show (纯净正文), fallback to show-content
    content_div = soup.find("div", class_="show")
    if not content_div or len(content_div.get_text(strip=True)) < 50:
        content_div = soup.find("div", class_="show-content")
    if not content_div or len(content_div.get_text(strip=True)) < 50:
        content_div = soup.find("div", class_="main_top_content")
    content = ""
    attachments = []

    if content_div:
        for tag in content_div.find_all(["script", "style"]):
            tag.decompose()

        parts = []
        has_table = 0
        link_seen = set()
        for el in content_div.find_all(["p", "div", "table", "img"]):
            # skip table-internal nodes (table itself handled as HTML)
            if el.name != "table" and el.find_parent("table"):
                continue
            # ⚠️ skip outer wrapper <p> that CONTAINS inner <p> (unclosed source HTML)
            #    otherwise it gets flattened as one giant paragraph → duplicate content
            if el.name == "p" and el.find("p", recursive=False):
                continue
            if el.name in ("p", "div"):
                # keep HTML when it contains attachments/links or images
                if el.find("a", href=True) or el.find("img"):
                    html_str = str(el)
                    # absolutize href/src
                    html_str = re.sub(
                        r'(<a\s[^>]*href=")([^"]+)"',
                        lambda m: m.group(1) + urljoin(url, m.group(2)) + '"',
                        html_str,
                    )
                    html_str = re.sub(
                        r'(<img\s[^>]*src=")([^"]+)"',
                        lambda m: m.group(1) + urljoin(url, m.group(2)) + '"',
                        html_str,
                    )
                    parts.append(html_str)
                else:
                    txt = el.get_text(" ", strip=True)
                    txt = re.sub(r"\s+", " ", txt)
                    if txt:
                        parts.append(txt)
            elif el.name == "table":
                has_table = 1
                parts.append(str(el))
            elif el.name == "img":
                src = el.get("src", "")
                if src and not src.startswith("data:"):
                    abs_src = urljoin(url, src)
                    parts.append('<img src="' + abs_src + '" alt="' + el.get("alt", "") + '" />')

        content = "\n\n".join(parts)

        # attachment links: absolutize + collect (embedded in content already via <a>)
        for a_tag in content_div.find_all("a", href=True):
            href = a_tag["href"]
            if re.search(r"\.(pdf|doc|docx|xls|xlsx|zip|rar)$", href, re.I) or "clientDf" in href:
                full_url = urljoin(url, href)
                name = a_tag.get_text(strip=True) or os.path.basename(href)
                if full_url not in link_seen:
                    link_seen.add(full_url)
                    attachments.append({"name": name, "url": full_url})
                # ensure embedded link present in content (normalize &amp; vs &)
                if full_url.replace("&", "&amp;") not in content and full_url not in content:
                    content += '\n\n<p><a href="' + full_url + '" target="_blank">' + name + "</a></p>"

    return title, date_str, attachments, content


def main():
    parser = argparse.ArgumentParser(description="台前县政府公告爬虫")
    parser.add_argument("--pages", type=int, default=5)
    args = parser.parse_args()

    pages = args.pages
    print("=== " + SITE_NAME + " ===")
    print("  pages: " + str(pages))

    all_items = []
    for page in range(1, pages + 1):
        if page == 1:
            url = LIST_URL
        else:
            url = "http://www.taiqian.gov.cn/channel/list/19256_" + str(page) + ".html"
        print("  [Page " + str(page) + "/" + str(pages) + "] fetching...", end=" ")
        try:
            html = fetch_page(url)
            items = parse_list_page(html)
            print(str(len(items)) + " items")
            all_items.extend(items)
        except Exception as e:
            print("ERROR: " + str(e))
            break

    print("\n=== Total " + str(len(all_items)) + " items ===")

    seen = set()
    unique = []
    for item in all_items:
        if item["url"] not in seen:
            seen.add(item["url"])
            unique.append(item)
    print("Unique: " + str(len(unique)))

    db_items = []
    for i, item in enumerate(unique):
        print("  [" + str(i + 1) + "/" + str(len(unique)) + "] " + item["title"][:40] + "...", end=" ")
        title, date_str, attachments, content = fetch_detail(item["url"])
        use_title = title or item["title"]
        use_date = date_str or item.get("date", "")
        summary = re.sub(r"\s+", "", content)[:200] if content else ""

        db_items.append({
            "site_name": SITE_NAME,
            "source_url": item["url"],
            "url": item["url"],
            "title": use_title,
            "pub_date": use_date,
            "summary": summary,
            "content": content,
            "attachments": json.dumps(attachments, ensure_ascii=False) if attachments else "",
        })
        print("ok [" + use_date + "]")

    try:
        from crawler_lib import push_to_searchdb
        push_to_searchdb(db_items, SITE_NAME)
        print("\nOK: " + str(len(db_items)))
    except Exception as e:
        print("FAIL: " + str(e))
        import traceback
        traceback.print_exc()


if __name__ == "__main__":
    main()
