#!/usr/bin/env python3
import os
"""
榆神工业区(榆林经济技术开发区) - 通知公告
WCM CMS, static HTML, index_N.html pagination
v2: fixed list parsing (li-based) + depth-counting content extraction
"""
import requests, re, sqlite3, time
from bs4 import BeautifulSoup

DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
BASE = "https://ysia.yl.gov.cn"
LIST_URL = BASE + "/xwzx/tzgg/index.html"
SITE_NAME = "榆神工业区(榆林经济技术开发区)"

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
}

inserted = 0


def extract_div_depth(html, start_pos):
    """Extract div content from start_pos using depth counting"""
    tag_end = html.index('>', start_pos)
    content_start = tag_end + 1

    depth = 1
    pos = content_start
    while pos < len(html) and depth > 0:
        next_open = html.find('<', pos)
        if next_open == -1:
            break

        # HTML comment
        if html[next_open:next_open + 4] == '<!--':
            close = html.find('-->', next_open + 4)
            if close == -1:
                break
            pos = close + 3
            continue

        close_tag_end = html.find('>', next_open)
        if close_tag_end == -1:
            break

        # Closing tag
        if html[next_open + 1] == '/':
            tag_name = html[next_open + 2:close_tag_end].split()[0].rstrip('>')
            if tag_name == 'div':
                depth -= 1
                if depth == 0:
                    return html[content_start:next_open].strip()
            pos = close_tag_end + 1
        else:
            tag_content = html[next_open + 1:close_tag_end]
            tag_name = tag_content.split()[0] if tag_content else ''

            # Check for self-closing
            if html[next_open:close_tag_end + 1].endswith('/>'):
                pos = close_tag_end + 1
                continue

            if tag_name == 'div':
                depth += 1

            pos = close_tag_end + 1

            # Skip script/style
            if tag_name in ('script', 'style'):
                close_tag = f'</{tag_name}>'
                sc_pos = html.find(close_tag, pos)
                if sc_pos != -1:
                    pos = sc_pos + len(close_tag)

    return html[content_start:pos].strip() if pos > content_start else ''


def get_article_list(page_url):
    """Parse article list - li-based, no m-lst36 dependency"""
    r = requests.get(page_url, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    soup = BeautifulSoup(r.text, 'html.parser')
    articles = []

    for li in soup.find_all('li'):
        a_tag = li.find('a')
        if not a_tag or not a_tag.get('href'):
            continue
        href = a_tag['href'].strip()
        if not href.endswith('.html') or href in ('./', '../', '../../'):
            continue

        title = a_tag.get('title', '').strip() or a_tag.get_text(strip=True)
        if not title or len(title) < 5:
            continue

        span_tag = li.find('span')
        date = span_tag.get_text(strip=True) if span_tag else ''

        if href.startswith('./'):
            href = BASE + '/xwzx/tzgg/' + href[2:]
        elif href.startswith('/'):
            href = BASE + href
        elif not href.startswith('http'):
            href = BASE + '/xwzx/tzgg/' + href

        articles.append({'title': title, 'url': href, 'date': date})

    return articles


def get_detail(detail_url):
    """Extract content with depth counting"""
    try:
        r = requests.get(detail_url, headers=HEADERS, timeout=30)
        r.encoding = 'utf-8'
        html = r.text
    except Exception as e:
        print(f"  [err] fetch: {e}")
        return None, None, None

    # Title from meta
    title = None
    mt = re.search(r'<meta[^>]*?ArticleTitle[^>]*?content="([^"]*?)"', html)
    if mt:
        title = mt.group(1).strip()
    if not title:
        return None, None, None
    title = re.sub(r'<[^>]+>', '', title).strip()

    # Content: depth-counting from TRS_UEDITOR div
    content = ''
    m = re.search(r'<div[^>]*class="[^"]*TRS_UEDITOR[^"]*"[^>]*>', html)
    if m:
        content = extract_div_depth(html, m.start())

    if not content:
        print(f"  [warn] No content: {title[:30]}")
        return title, '', None

    # Publish date from meta
    pub_date = None
    mt3 = re.search(r'<meta[^>]*?PubDate[^>]*?content="([^"]+?)"', html, re.IGNORECASE)
    if mt3:
        pub_date = mt3.group(1)[:10]

    return title, content, pub_date


def save_article(conn, title, content, pub_date, page_url, source_url):
    global inserted
    if not title or not pub_date:
        return
    conn.execute(
        """INSERT OR REPLACE INTO gov_raw (title, content, summary, site_name, source_url, page_url, publish_date, category, script_name) VALUES (?,?,?,?,?,?,?,?, 'crawl_ysia.py')""",
        (title, content, content[:500] if content else "", SITE_NAME,
         source_url or SITE_NAME, page_url, pub_date, "通知公告")
    )
    inserted += 1


def main():
    global inserted
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=5000")

    print("[ysia] Fetching list...")
    r = requests.get(LIST_URL, headers=HEADERS, timeout=30)
    r.encoding = 'utf-8'
    m = re.search(r'createPage\((\d+)', r.text)
    total_pages = int(m.group(1)) if m else 32
    print(f"[ysia] Total pages: {total_pages}")

    all_articles = []
    for page in range(total_pages - 1):  # 0 to N-1
        if page == 0:
            page_url = LIST_URL
        else:
            page_url = f"{BASE}/xwzx/tzgg/index_{page}.html"

        arts = get_article_list(page_url)
        all_articles.extend(arts)
        print(f"[ysia] Page {page + 1}: {len(arts)} articles (total: {len(all_articles)})")
        time.sleep(0.3)

    if not all_articles:
        print("[ysia] No articles found!")
        conn.close()
        return

    for idx, art in enumerate(all_articles):
        title, content, pub_date = get_detail(art['url'])
        if title and content is not None:
            if not pub_date:
                pub_date = art['date']
            save_article(conn, title, content, pub_date, art['url'], art['url'])

        if (idx + 1) % 20 == 0:
            conn.commit()
            print(f"[ysia] Progress: {idx + 1}/{len(all_articles)} ({inserted} saved)")
        time.sleep(0.2)

    conn.commit()
    print(f"[ysia] Done: {inserted} saved (of {len(all_articles)} total)")
    conn.close()


if __name__ == "__main__":
    main()
