#!/usr/bin/env python3
"""
Crawler: 鞍山市行政审批局 - 通知公告 (TRS WCM static HTML)
List: glist.html (page 1), glistN.html (page N+1), 101 pages x 20 = ~2015 items
Detail: /html/ASSPJ/YYYYMM/017xxxxxxxxxxxxx.html
Content div: hwq-info-article-center
"""
import sys, os, json, re, time, requests
from datetime import datetime, timedelta
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
SITE = "鞍山市行政审批局"
COLUMN = "通知公告"
BASE_URL = "http://xzspj.anshan.gov.cn"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/125.0.0.0 Safari/537.36",
}
ATTACH_EXTS = ('.doc', '.docx', '.pdf', '.xls', '.xlsx', '.ppt', '.pptx', '.zip', '.rar', '.7z')
MAX_PAGES = 101  # 总101页


def log(msg):
    print(f"[{SITE}] {msg}", flush=True)


def fetch_list_page(page_num):
    """Fetch list page. page 1 = glist.html, page N+1 = glistN.html"""
    if page_num == 1:
        url = f"{BASE_URL}/assxzspj/zwzx/tzgg/glist.html"
    else:
        url = f"{BASE_URL}/assxzspj/zwzx/tzgg/glist{page_num - 1}.html"
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        log(f"List page {page_num} error: {e}")
        return None


def parse_list_page(html):
    """Extract (title, url, date) from list page."""
    items = []
    # Find all <div class="name"><a href="URL">TITLE</a></div> with <div class="time">DATE</div> before them
    # Pattern: <div class="time">DATE</div> ... <div class="name"><a href="URL">TITLE</a></div>
    # Extract pairs
    soup = BeautifulSoup(html, "html.parser")
    # Find the ul with the list
    ul = soup.find("ul", style=lambda v: v and "display" in v if v else False)
    if not ul:
        # fallback: find all div.info > ul
        info_div = soup.find("div", class_=lambda c: c and "info" in c.split())
        if info_div:
            ul = info_div.find("ul")
    
    if ul:
        lis = ul.find_all("li", recursive=False)
        for li in lis:
            time_div = li.find("div", class_=lambda c: c and "time" in c.split() if c else False)
            name_div = li.find("div", class_=lambda c: c and "name" in c.split() if c else False)
            if time_div and name_div:
                date = time_div.get_text(strip=True)
                a_tag = name_div.find("a")
                if a_tag and a_tag.get("href"):
                    title = a_tag.get_text(strip=True)
                    url = a_tag["href"]
                    items.append((title, url, date))
    return items


def fetch_detail(detail_url):
    """Fetch detail page."""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        log(f"Detail error {detail_url}: {e}")
        return None


def parse_detail(html, detail_url):
    """Extract title, date, content from detail page."""
    result = {"title": "", "publish_date": "", "content": ""}
    soup = BeautifulSoup(html, "html.parser")

    # Title
    title_div = soup.find("div", class_=lambda c: c and "hwq-info-article-title" in c.split() if c else False)
    if title_div:
        result["title"] = title_div.get_text(strip=True)
    if not result["title"]:
        m = re.search(r'<title>(.*?)</title>', html)
        if m:
            result["title"] = m.group(1).strip()

    # Date from time div
    time_div = soup.find("div", class_=lambda c: c and "hwq-info-article-time" in c.split() if c else False)
    if time_div:
        time_text = time_div.get_text()
        m = re.search(r'(\d{4}-\d{2}-\d{2})', time_text)
        if m:
            result["publish_date"] = m.group(1)

    # Content
    content_div = soup.find("div", class_=lambda c: c and "hwq-info-article-center" in c.split() if c else False)
    if content_div:
        result["content"] = extract_clean_text(str(content_div))
    else:
        # Fallback: get anything
        body = soup.find("body")
        if body:
            result["content"] = body.get_text(separator="\n", strip=True)

    return result


def extract_clean_text(content_html):
    """Convert content HTML to text with markdown links, preserving tables."""
    if not content_html:
        return ""
    soup = BeautifulSoup(content_html, "html.parser")

    # Convert attachment links to markdown
    for a_tag in soup.find_all("a", href=True):
        href = a_tag.get("href", "")
        text = a_tag.get_text(strip=True) or "附件"
        full_url = urljoin(BASE_URL, href)
        a_tag.replace_with(f"[{text}]({full_url})")

    # Unwrap inline formatting tags
    for tag in soup.find_all(['span', 'b', 'strong', 'font', 'em', 'i', 'u', 's']):
        tag.unwrap()

    parts = []
    for el in soup.find_all(['table', 'p']):
        if el.name == 'table':
            parts.append(str(el))
        elif el.name == 'p' and not el.find_parent('table'):
            t = el.get_text(separator='', strip=True)
            if t:
                parts.append(t)
    return '\n\n'.join(parts) if parts else content_html.strip()


def crawl_all(months_back=36):
    cutoff = (datetime.now() - timedelta(days=months_back * 30)).strftime("%Y-%m-%d")
    log(f"Full crawl (cutoff: {cutoff})")

    all_items = []
    for pg in range(1, MAX_PAGES + 1):
        html = fetch_list_page(pg)
        if not html:
            break
        items = parse_list_page(html)
        if not items:
            break
        for title, url, date in items:
            if date < cutoff:
                continue
            # Skip non-article links (navigation)
            if not url or "glist" in url.lower() or url == "#":
                continue
            all_items.append({
                "title": title,
                "url": url,
                "date": date,
            })
        log(f"  Page {pg}: {len(items)} items (accumulated: {len(all_items)})")
        time.sleep(0.5)
        # Check if we've reached the end (fewer items on last page)
        if len(items) < 20:
            break

    log(f"After filter: {len(all_items)} items")

    # Fetch details
    enriched = []
    for idx, item in enumerate(all_items, 1):
        html = fetch_detail(item["url"])
        if html:
            detail = parse_detail(html, item["url"])
            if detail["title"]:
                item["title"] = detail["title"]
            if detail["publish_date"]:
                item["date"] = detail["publish_date"]
            item["content"] = detail["content"]
        else:
            item["content"] = ""

        enriched.append({
            "title": item["title"],
            "page_url": item["url"],
            "publish_date": item["date"],
            "content": item.get("content", ""),
            "site_name": f"{SITE}-{COLUMN}",
            "column": COLUMN,
        })
        if idx % 20 == 0:
            log(f"  Detail: {idx}/{len(all_items)}")
        time.sleep(0.5)

    # Store
    stored = 0
    skipped = 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], item["column"]))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


def crawl_incremental():
    cutoff = (datetime.now() - timedelta(days=7)).strftime("%Y-%m-%d")
    log(f"Incremental (since {cutoff})")

    # Only check first page for new items
    html = fetch_list_page(1)
    if not html:
        return
    items = parse_list_page(html)
    filtered = [(t, u, d) for t, u, d in items if d >= cutoff and not u.startswith("http") and "glist" not in u]

    enriched = []
    for title, url, date in filtered:
        html = fetch_detail(url)
        detail = parse_detail(html, url) if html else {"title": "", "publish_date": "", "content": ""}
        enriched.append({
            "title": detail["title"] or title,
            "page_url": url,
            "publish_date": date,
            "content": detail["content"],
            "site_name": f"{SITE}-{COLUMN}",
            "column": COLUMN,
        })
        time.sleep(0.5)

    stored = 0
    skipped = 0
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=10000")
    c = conn.cursor()
    for item in enriched:
        try:
            c.execute(
                "INSERT OR IGNORE INTO gov_raw "
                "(site_name, source_url, page_url, title, publish_date, content, summary, category)"
                " VALUES (?,?,?,?,?,?,?,?)",
                (item["site_name"], item["page_url"], item["page_url"],
                 item["title"], item["publish_date"], item["content"],
                 (item["content"] or "")[:500], item["column"]))
            if c.rowcount > 0:
                stored += 1
            else:
                skipped += 1
        except Exception as e:
            log(f"DB error: {e}")
            skipped += 1
    conn.commit()
    conn.close()
    log(f"Result: {stored} new, {skipped} skipped")


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        crawl_all()
    elif mode == "list":
        html = fetch_list_page(1)
        items = parse_list_page(html) if html else []
        log(f"Page 1: {len(items)} items")
        for t, u, d in items[:5]:
            print(f"  {d} | {t[:60]}")
