#!/usr/bin/env python3
"""Crawl gzxw.com.cn - 甘州在线 公告栏目(sid=384)"""
import re, sys, sqlite3, urllib.request, urllib.parse, traceback
from datetime import datetime

BASE_URL = "https://www.gzxw.com.cn"
SITE_NAME = "gzxxw"
DB_PATH = "/root/search.db"
MAX_PAGES = 20

headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}

def fetch(url):
    req = urllib.request.Request(url, headers=headers)
    resp = urllib.request.urlopen(req, timeout=15)
    return resp.read().decode("utf-8", errors="replace")

def get_items_from_page(page_num):
    """Get all (title, url, id) from a list page"""
    if page_num == 1:
        url = f"{BASE_URL}/index.php/index/news/newslist.html?sid=384"
    else:
        url = f"{BASE_URL}/index.php/index/news/newslist.html?sid=384&page={page_num}"
    html = fetch(url)
    items = set()
    pattern = re.compile(r'href="([^"]*detail_news[^"]*id=(\d+)[^"]*)"[^>]*>([^<]+)<')
    matches = pattern.findall(html)
    seen_ids = set()
    for href, id_val, text in matches:
        text = text.strip()
        if id_val not in seen_ids and text and len(text) > 5:
            seen_ids.add(id_val)
            full_url = href if href.startswith("http") else BASE_URL + href
            items.add((text, full_url, id_val))
    page_links = re.findall(r'href="[^"]*page=(\d+)[^"]*"', html)
    max_page = max([int(p) for p in page_links] + [page_num])
    return items, max_page


def extract_clean_text(html_chunk):
    """Extract text from HTML chunk, preserving paragraph breaks."""
    text = re.sub(r"</p>", "\n\n", html_chunk, flags=re.I)
    text = re.sub(r"<br\s*/?>", "\n", text, flags=re.I)
    text = re.sub(r"</div>", "\n\n", text, flags=re.I)
    text = re.sub(r"</?(?:b|span|font|strong|em|u|i|a)\b[^>]*>", "", text, flags=re.I)
    text = re.sub(r"<p\b[^>]*>", "", text, flags=re.I)
    text = re.sub(r"\n{3,}", "\n\n", text)
    lines = [l.strip() for l in text.split("\n")]
    text = "\n".join(lines)
    text = re.sub(r"\n{3,}", "\n\n", text)
    text = text.strip()
    return text


def extract_attachments(html):
    """Extract attachment links from div.attachment_url, return as markdown string."""
    parts = []
    idx = html.find('class="attachment_url"')
    if idx < 0:
        return ""
    start = html.rfind("<div", 0, idx)
    if start < 0:
        start = idx
    pos = html.find(">", idx) + 1
    depth = 1
    end = pos
    while depth > 0 and end < len(html):
        next_open = html.find("<div", end)
        next_close = html.find("</div>", end)
        if next_close < 0:
            break
        if next_open >= 0 and next_open < next_close:
            depth += 1
            end = html.find(">", next_open) + 1
        else:
            depth -= 1
            end = next_close + 6
    chunk = html[pos:end-6] if depth == 0 else html[pos:]
    links = re.findall(r'<a\s[^>]*href="([^"]+)"[^>]*>([^<]+)</a>', chunk)
    for href, name in links:
        full_url = href if href.startswith("http") else BASE_URL + href
        parts.append(f"[{name.strip()}]({full_url})")
    return "\n".join(parts)


def get_detail(url):
    """Get (title, date, content, source) from detail page"""
    html = fetch(url)
    title_match = re.search(r"<title>([^<]+)</title>", html)
    title = title_match.group(1).strip() if title_match else ""
    title = re.sub(r"\s*[-–—|_].*$", "", title).strip()

    content = ""
    idx = html.find('class="com_text"')
    if idx > 0:
        start = html.rfind("<div", 0, idx)
        if start < 0:
            start = idx
        pos = html.find(">", idx) + 1
        depth = 1
        end = pos
        while depth > 0 and end < len(html):
            next_open = html.find("<div", end)
            next_close = html.find("</div>", end)
            if next_close < 0:
                break
            if next_open >= 0 and next_open < next_close:
                depth += 1
                end = html.find(">", next_open) + 1
            else:
                depth -= 1
                end = next_close + 6
        chunk = html[pos:end-6] if depth == 0 else html[pos:]
        content = extract_clean_text(chunk)

    attachments = extract_attachments(html)
    if attachments:
        if content:
            content += "\n\n"
        content += attachments

    date = ""
    idx2 = html.find('class="content_box"')
    if idx2 > 0:
        chunk = html[idx2:idx2+2000]
        bar_match = re.search(r'class="top-bar"[^>]*>.*?<span>(\d{4}-\d{2}-\d{2})', chunk)
        if bar_match:
            date = bar_match.group(1)
        else:
            dates = re.findall(r"(\d{4}-\d{2}-\d{2})", chunk)
            if dates:
                date = dates[0]

    source_match = re.search(r"来源[：:]\s*([^<]+)", html)
    source = source_match.group(1).strip() if source_match else ""
    return title, date, content, source


def save_to_db(records):
    if not records:
        return 0
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA busy_timeout=30000")
    c = conn.cursor()
    count = 0
    for r in records:
        existing = c.execute("SELECT id FROM gov_raw WHERE source_url=?", (r["url"],)).fetchone()
        if existing:
            print(f"  SKIP: {r['title'][:40]}")
            continue
        c.execute("""
            INSERT INTO gov_raw (site_name, source_url, page_url, title, publish_date, date_rank, summary, content)
            VALUES (?, ?, ?, ?, ?, ?, ?, ?)
        """, (
            SITE_NAME, r["url"], r["url"], r["title"],
            r["date"],
            int(datetime.strptime(r["date"], "%Y-%m-%d").timestamp()) if r["date"] else 0,
            r["content"][:200] if r["content"] else "",
            r["content"],
        ))
        conn.commit()
        count += 1
        print(f"  OK: {r['title'][:50]} | {r['date']}")
    conn.close()
    return count


def main():
    print("=== Crawling gzxw.com.cn (公告 sid=384) ===")
    all_items = []
    max_page = 1
    for page_num in range(1, MAX_PAGES + 1):
        print(f"\nPage {page_num}...")
        try:
            items, max_p = get_items_from_page(page_num)
            max_page = max(max_page, max_p)
            if not items:
                print(f"  No items, stopping")
                break
            print(f"  Found {len(items)} items (max_page={max_p})")
            all_items.extend(items)
            if page_num >= max_page:
                break
        except Exception as e:
            print(f"  Error page {page_num}: {e}")
            break
    print(f"\n=== Total: {len(all_items)} items ===")
    seen = set()
    unique_items = []
    for title, url, id_val in all_items:
        if id_val not in seen:
            seen.add(id_val)
            unique_items.append((title, url, id_val))
    print(f"Unique: {len(unique_items)} items")
    records = []
    for i, (title, url, id_val) in enumerate(unique_items):
        print(f"\n[{i+1}/{len(unique_items)}] {title[:50]}...")
        try:
            d_title, d_date, content, source = get_detail(url)
            records.append({"title": d_title or title, "url": url, "date": d_date, "content": content, "source": source})
        except Exception as e:
            print(f"  ERROR: {e}")
            traceback.print_exc()
    saved = save_to_db(records)
    print(f"\n=== Complete: {saved}/{len(records)} new records ===")

if __name__ == "__main__":
    main()
