#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""芜湖市生态环境局 - 公示公告/征集 (Lonsun CMS)"""
import json, re, sys, time, requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "芜湖市生态环境局-公示征集"
CATEGORY = "安徽"
GROUP = "安徽"
BASE_URL = "https://sthjj.wuhu.gov.cn"
COLUMN_ID = 6788491
COLUMN_URL = "https://sthjj.wuhu.gov.cn/content/column/%d" % COLUMN_ID
MAX_PAGES = 5
PER_PAGE = 20

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

INSERT_SQL = """INSERT OR IGNORE INTO gov_raw
    (site_name, source_url, page_url, title, publish_date, summary, content, category, attachments, group_name)
    VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)"""


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    import urllib.parse
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)
def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def extract_list_items(html_text):
    """Extract (title, url, date) from list page HTML."""
    soup = BeautifulSoup(html_text, "html.parser")
    items = []
    ul = soup.find("ul", class_=re.compile(r"doc_list"))
    if not ul:
        return items
    for li in ul.find_all("li", recursive=False):
        a = li.find("a")
        if not a:
            continue
        title = (a.get("title") or a.get_text(strip=True) or "").strip()
        if not title:
            continue
        href = a.get("href", "").strip()
        if not href:
            continue
        full_url = href if href.startswith("http") else urljoin(BASE_URL, href)
        span = li.find("span", class_="date")
        date = span.get_text(strip=True) if span else ""
        items.append({"title": title, "url": full_url, "date": date})
    return items


def extract_detail(url):
    """Extract title, date, content, attachments from a detail page."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        if r.status_code != 200:
            return None, None, None, None
    except Exception as e:
        print(f"  [ERROR] fetch detail failed: {e}", file=sys.stderr)
        return None, None, None, None

    soup = BeautifulSoup(r.text, "html.parser")

    # --- Title ---
    title = ""
    meta_title = soup.find("meta", attrs={"name": "ArticleTitle"})
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()
    if not title:
        h1 = soup.find("h1")
        if h1:
            title = h1.get_text(strip=True)

    # --- Date ---
    pub_date = ""
    meta_date = soup.find("meta", attrs={"name": "PubDate"})
    if meta_date and meta_date.get("content"):
        pub_date = meta_date["content"].strip()[:10]

    # --- Content ---
    content_parts = []
    attachments = []
    content_div = soup.find("div", class_="newscontnet")
    if content_div:
        for noise in content_div.find_all(["script", "style"]):
            noise.decompose()

        # Check if it's images-only content
        imgs = content_div.find_all("img")
        has_text = False
        for p in content_div.find_all("p"):
            txt = p.get_text(" ", strip=True)
            if txt and len(txt) > 5:
                has_text = True
                break

        if imgs and not has_text:
            # Images-only: embed as markdown
            img_lines = ["[%s](%s)" % (title or url.split("/")[-1], url)]
            for img in imgs:
                src = img.get("src", "")
                if src:
                    full_src = src if src.startswith("http") else urljoin(url, src)
                    alt = img.get("alt", "") or img.get("title", "") or ""
                    img_lines.append("![%s](%s)" % (alt, full_src))
            content_text = "\n\n".join(img_lines)
            summary = content_text[:200]
            return title, pub_date, content_text, "[]"
        else:
            # Normal text content
            for child in content_div.find_all(["p", "table"], recursive=True):
                if child.name == 'table':
                    tbl_html = html_table_to_html(child, url)
                    if tbl_html:
                        content_parts.append(tbl_html)
                elif child.name == "p":
                    txt = child.get_text(" ", strip=True)
                    if txt:
                        # Handle inline images in p
                        p_imgs = child.find_all("img")
                        if p_imgs and not txt.strip():
                            for img in p_imgs:
                                src = img.get("src", "")
                                if src:
                                    full_src = src if src.startswith("http") else urljoin(url, src)
                                    alt = img.get("alt", "") or ""
                                    content_parts.append("![%s](%s)" % (alt, full_src))
                        else:
                            content_parts.append(txt)

            # --- Attachments ---
            for a in content_div.find_all("a", href=True):
                href = a["href"].strip()
                if re.search(r'\.(pdf|doc|docx|xls|xlsx|zip|rar)$', href, re.I):
                    full_href = href if href.startswith("http") else urljoin(url, href)
                    attachments.append({
                        "name": a.get_text(strip=True) or href.split("/")[-1],
                        "url": full_href,
                    })

    if not content_parts:
        content_parts.append("[%s](%s)" % (title or url.split("/")[-1], url))

    content_text = "\n\n".join(content_parts)
    summary = content_text[:200] if content_text else title

    return title, pub_date, content_text, json.dumps(attachments, ensure_ascii=False)


def crawl(test_mode=False, max_pages=MAX_PAGES):
    conn = init_db()
    cur = conn.cursor()
    total = 0
    errors = 0

    for pn in range(1, max_pages + 1):
        if pn == 1:
            list_url = "https://sthjj.wuhu.gov.cn/hbzx/gszq/index.html"
        else:
            list_url = COLUMN_URL + "?pageIndex=%d" % pn

        print(f"[Page {pn}] {list_url}", file=sys.stderr)

        try:
            r = requests.get(list_url, headers=HEADERS, timeout=30)
            r.encoding = "utf-8"
            if r.status_code != 200:
                print(f"  [ERROR] list page {pn} HTTP {r.status_code}", file=sys.stderr)
                errors += 1
                continue
        except Exception as e:
            print(f"  [ERROR] list page {pn}: {e}", file=sys.stderr)
            errors += 1
            continue

        items = extract_list_items(r.text)
        print(f"  Found {len(items)} items", file=sys.stderr)

        if not items:
            print(f"  [WARN] No items found, stopping", file=sys.stderr)
            break

        for item in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (item["url"],))
            if cur.fetchone():
                print(f"  SKIP (exists): {item['title'][:40]}", file=sys.stderr)
                continue

            title, pub_date, content, attachments = extract_detail(item["url"])
            if title is None:
                print(f"  [ERROR] detail unreachable: {item['url']}", file=sys.stderr)
                errors += 1
                continue

            final_title = title or item["title"]
            final_date = pub_date or item["date"]
            final_summary = content[:200] if content else final_title

            if test_mode:
                print(f"  [{final_date}] {final_title[:50]}", file=sys.stderr)
                total += 1
                continue

            try:
                cur.execute(INSERT_SQL, (
                    SITE_NAME, item["url"], item["url"],
                    final_title.strip(), final_date,
                    final_summary.strip(), content or ("[%s](%s)" % (final_title, item["url"])),
                    CATEGORY, attachments, GROUP,
                ))
                conn.commit()
                total += 1
                print(f"  OK: {final_title[:40]}", file=sys.stderr)
            except Exception as e:
                print(f"  DB error: {e}", file=sys.stderr)
                errors += 1

        time.sleep(1)

    conn.close()
    print(f"\n[DONE] Total: {total}, Errors: {errors}", file=sys.stderr)
    return total


if __name__ == "__main__":
    test_mode = "--test" in sys.argv
    mp = MAX_PAGES
    for i, a in enumerate(sys.argv):
        if a.startswith("--max-pages="):
            mp = int(a.split("=")[1])
    crawl(test_mode=test_mode, max_pages=mp)
