#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
营口经济技术开发区（鲅鱼圈区）- 通知公告 爬虫
CMS: webBuilder
List: /003/003006/about.html (P1), /003/003006/{N}.html (P2-P6)
Detail: /003/003006/YYYYMMDD/uuid.html
每页约16条
"""

import requests, re, json, sqlite3, time, os, sys
from datetime import datetime
from bs4 import BeautifulSoup

DB_PATH = "/root/search.db"
SITE_NAME = "营口经济技术开发区-通知公告"
CATEGORY = "通知公告"
GROUP = "辽宁"
BASE_URL = "http://www.ykdz.gov.cn"
MAX_PAGES = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

SKIP_TEXTS = [
    "【打印本页】", "【关闭窗口】", "打印本页", "关闭窗口",
    "转载分享：", "浏览量：", "相关附件：", "相关稿件：",
    "扫一扫在手机打开", "附件：",
]

# ---------- Table to Markdown ----------

def table_to_md(table, *args, **kwargs):
    """保留 HTML 表格结构（不转 md）"""
    return str(table)

def render_paragraph(p_tag, detail_url):
    """Render a <p> preserving <a> links and <img> as markdown."""
    parts = []
    for elem in p_tag.contents:
        if isinstance(elem, str):
            t = elem.strip()
            if t:
                parts.append(t)
        elif elem.name == "a":
            href = elem.get("href", "").strip()
            text = elem.get_text(strip=True)
            if href and text:
                full = href if href.startswith("http") else (BASE_URL + href if href.startswith("/") else BASE_URL + "/" + href.lstrip("/"))
                parts.append(f"[{text}]({full})")
            elif text:
                parts.append(text)
        elif elem.name == "img":
            src = elem.get("src", "")
            if src:
                alt = elem.get("alt", "")
                full_src = src if src.startswith("http") else (BASE_URL + src if src.startswith("/") else BASE_URL + "/" + src.lstrip("/"))
                parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
        elif hasattr(elem, "get_text"):
            t = elem.get_text(" ", strip=True)
            if t:
                parts.append(t)
    return " ".join(parts)


def extract_content(content_div, detail_url):
    """Extract content from content_div: paragraphs + tables + images."""
    parts = []
    processed = set()

    # Remove style/script
    for s in content_div.find_all(["script", "style"]):
        s.decompose()

    for child in content_div.children:
        if child.name is None:
            continue
        tn = child.name.lower()

        if tn == "p":
            txt = render_paragraph(child, detail_url)
            if txt:
                parts.append(txt)

        elif tn == "table":
            all_text = child.get_text(strip=True)
            if len(all_text) < 15:
                continue
            md = table_to_md(child)
            if md:
                parts.append(md)

        elif tn == "img":
            src = child.get("src", "")
            if src and id(child) not in processed:
                processed.add(id(child))
                alt = child.get("alt", "")
                full_src = src if src.startswith("http") else (BASE_URL + src if src.startswith("/") else BASE_URL + "/" + src.lstrip("/"))
                parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")

        elif tn in ("div", "section", "center"):
            # Recursively handle sub-elements
            for sub in child.find_all(["p", "table", "img"], recursive=True):
                if id(sub) in processed:
                    continue
                processed.add(id(sub))
                if sub.name == "p":
                    txt = render_paragraph(sub, detail_url)
                    if txt:
                        parts.append(txt)
                elif sub.name == "table":
                    all_text = sub.get_text(strip=True)
                    if len(all_text) < 15:
                        continue
                    md = table_to_md(sub)
                    if md:
                        parts.append(md)
                elif sub.name == "img":
                    src = sub.get("src", "")
                    if src:
                        alt = sub.get("alt", "")
                        full_src = src if src.startswith("http") else (BASE_URL + src if src.startswith("/") else BASE_URL + "/" + src.lstrip("/"))
                        parts.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")

        elif tn in ("ul", "ol"):
            for li in child.find_all("li", recursive=False):
                txt = li.get_text(" ", strip=True)
                if txt:
                    parts.append("- " + txt)

    # Filter out noise lines
    filtered = []
    for p in parts:
        if any(sk in p for sk in SKIP_TEXTS):
            continue
        if re.match(r'^(来源|发布日期)[：:]\s*', p):
            continue
        filtered.append(p)

    return "\n\n".join(filtered)


# ---------- Attachments ----------

def extract_attachments(soup):
    """Find PDF/doc/xls/xlsx/zip/rar links in the page."""
    atts = []
    seen = set()
    for a in soup.find_all("a", href=True):
        href = a["href"]
        if re.search(r"\.(pdf|doc|docx|xls|xlsx|rar|zip|txt)$", href, re.I):
            if href.startswith("/"):
                href = BASE_URL + href
            elif not href.startswith("http"):
                href = BASE_URL + "/" + href.lstrip("/")
            name = a.get_text(strip=True) or href.split("/")[-1].split("?")[0]
            if href not in seen:
                seen.add(href)
                atts.append({"name": name, "url": href})
    return atts


# ---------- Detail page ----------

def fetch_detail(url):
    """Fetch detail page: extract content, title, date, attachments."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        return None, None, None, []

    soup = BeautifulSoup(r.text, "html.parser")

    # --- Title ---
    # Prefer h3 in content area, fallback to <title>
    title = ""
    h3 = soup.find("h3")
    if h3:
        t = h3.get_text(strip=True)
        if t and len(t) > 5 and "通知公告" not in t:
            title = t
    if not title:
        title_tag = soup.find("title")
        if title_tag:
            t = title_tag.get_text(strip=True)
            # Clean suffix
            t = re.sub(r"-营口市鲅鱼圈区人民政府、营口经济技术开发区$", "", t).strip()
            title = t
    if not title:
        # Fallback: from URL list
        return None, None, None, []

    # --- Date ---
    date_str = ""
    dm = re.search(r"发布时间[：:]\s*(\d{4}-\d{1,2}-\d{1,2})", r.text)
    if dm:
        date_str = dm.group(1)
    if not date_str:
        # Try URL date
        um = re.search(r"/(\d{8})/", url)
        if um:
            d = um.group(1)
            date_str = f"{d[:4]}-{d[4:6]}-{d[6:8]}"

    # --- Content ---
    content_div = soup.find("div", class_="ewb-article-content")
    if not content_div:
        content_div = soup.find("div", id="ivs_content")

    attachments = extract_attachments(soup)

    if content_div:
        content = extract_content(content_div, url)
    else:
        content = ""

    # Fallback: image-only page
    if len(content.strip()) < 20:
        content = f'<p><a href="{url}">{title}</a></p>'
        imgs = soup.find_all("img")
        img_lines = []
        for img in imgs:
            src = img.get("src", "")
            if src:
                full_src = src if src.startswith("http") else (BASE_URL + src if src.startswith("/") else BASE_URL + "/" + src.lstrip("/"))
                alt = img.get("alt", "")
                img_lines.append(f"![{alt}]({full_src})" if alt else f"![]({full_src})")
        if img_lines:
            content += "\n\n" + "\n\n".join(img_lines)

    if attachments:
        att_lines = []
        for a in attachments:
            att_lines.append(f"附件：[{a['name']}]({a['url']})")
        content += "\n\n" + "\n\n".join(att_lines)

    return title, date_str, content, attachments


# ---------- List page ----------

def extract_list_page(page_num):
    """Extract article list from a given page number."""
    if page_num == 1:
        url = f"{BASE_URL}/003/003006/about.html"
    else:
        url = f"{BASE_URL}/003/003006/{page_num}.html"

    try:
        r = requests.get(url, headers=HEADERS, timeout=20)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  Page {page_num} error: {e}")
        return []

    soup = BeautifulSoup(r.text, "html.parser")

    er = soup.find("div", class_="ewb-right")
    if not er:
        return []

    articles = []
    for ul in er.find_all("ul"):
        lis = ul.find_all("li", class_="ewb-list-node")
        if len(lis) >= 3:
            for li in lis:
                a = li.find("a")
                if not a:
                    continue
                href = a.get("href", "")
                # Filter article URLs
                if not re.search(r"/003/003006/\d{8}/[a-f0-9-]+\.html$", href):
                    continue

                # Title: prefer title attribute (complete)
                title = a.get("title", "") or a.get_text(strip=True)
                if not title or len(title) < 5:
                    continue

                # Full URL
                if href.startswith("/"):
                    full_url = BASE_URL + href
                elif href.startswith("http"):
                    full_url = href
                else:
                    full_url = BASE_URL + "/" + href.lstrip("/")

                # Date from URL path
                date = ""
                dm = re.search(r"/(\d{8})/", href)
                if dm:
                    d = dm.group(1)
                    date = f"{d[:4]}-{d[4:6]}-{d[6:8]}"

                articles.append((title, full_url, date))
            break  # Only the first relevant ul

    return articles


# ---------- DB ----------

def init_db():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def save_article(conn, title, page_url, publish_date, content, attachments):
    summary = content[:200] if content else title
    summary = re.sub(r"\s+", " ", summary).strip()
    att_json = json.dumps(attachments, ensure_ascii=False) if attachments else "[]"
    try:
        conn.execute(
            """INSERT OR IGNORE INTO gov_raw
               (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
               VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""",
            (SITE_NAME, page_url, page_url, title.strip(), publish_date,
             summary, content, CATEGORY, att_json, GROUP),
        )
        return conn.total_changes > 0
    except Exception as e:
        print(f"  DB error: {e}", file=sys.stderr)
        return False


# ---------- Main ----------

def main():
    max_pages = MAX_PAGES
    if len(sys.argv) > 1:
        try:
            max_pages = int(sys.argv[1])
        except ValueError:
            if sys.argv[1] in ("--max-pages",):
                if len(sys.argv) > 2:
                    try:
                        max_pages = int(sys.argv[2])
                    except ValueError:
                        pass

    print(f"Scraping {SITE_NAME}, max pages: {max_pages}")

    conn = init_db()
    total_new = 0
    total_found = 0

    for pn in range(1, max_pages + 1):
        articles = extract_list_page(pn)
        if not articles:
            print(f"  Page {pn}: 0 articles (end)")
            break

        total_found += len(articles)

        for title, page_url, date in articles:
            # Skip very old articles
            if date and date < "2020-01-01":
                continue

            # Check if already exists
            cur = conn.execute("SELECT id FROM gov_raw WHERE page_url = ?", (page_url,))
            if cur.fetchone():
                continue

            det_title, det_date, det_content, det_atts = fetch_detail(page_url)
            final_title = det_title or title
            final_date = det_date or date
            final_content = det_content or f"[{final_title}]({page_url})"
            final_atts = det_atts or []

            if save_article(conn, final_title, page_url, final_date, final_content, final_atts):
                total_new += 1

            # Brief delay
            time.sleep(0.3)

        print(f"  Page {pn}: {len(articles)} found, {total_new} new so far")

    conn.commit()
    conn.close()
    print(f"\nDone! Total found: {total_found}, New: {total_new}")


if __name__ == "__main__":
    main()
