#!/usr/bin/env python3
"""
郴州新网 - 公告头条 (0735.com)
https://www.0735.com/toutiao/list-22.html
本地生活信息门户 - 自定义CMS
"""
import re
import sys
import time
import json
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

DB_PATH = "/root/search.db"
SITE_NAME = "郴州新网-公告头条"
CATEGORY = "企业"
BASE_URL = "https://www.0735.com/toutiao/list-22.html"
MAX_PAGES = 13  # 共13页, 506条

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
    "Referer": "https://www.0735.com/toutiao/",
}

INSERT_SQL = """
INSERT OR IGNORE INTO gov_raw 
    (site_name, page_url, title, publish_date, summary, content, category, attachments)
    VALUES (?, ?, ?, ?, '', ?, ?, ?)
"""


# ─── 正文取文本（2026-09-11）：行内节点直接拼接，只在块级边界 / <br> 处换行 ───
# ⚠️ 不要用 el.get_text("\n") 取正文 —— 它是「每个**文本节点**之间插 \n」，Word 粘贴的
#    公文把一行拆成 <span>提取码：</span>pwaj<span>。查阅…</span>，这些行内节点于是各自
#    成行（福泉 id=2095080103703914437 实例：`提取码：`/`pwaj`/`。查阅…` 各占一行）。
_BLOCK_TAGS = {'address', 'article', 'aside', 'blockquote', 'details', 'dialog', 'dd', 'div',
               'dl', 'dt', 'fieldset', 'figcaption', 'figure', 'footer', 'form', 'h1', 'h2',
               'h3', 'h4', 'h5', 'h6', 'header', 'hgroup', 'hr', 'li', 'main', 'nav', 'ol',
               'p', 'pre', 'section', 'table', 'tbody', 'thead', 'tfoot', 'tr', 'td', 'th',
               'ul', 'center', 'caption'}


def body_text(el):
    """块级边界出换行、行内节点直接拼接、<br> 出换行（≈ 浏览器看到的换行结构）。"""
    if el is None:
        return ''
    import re as _re
    from bs4 import NavigableString
    out = []

    def walk(node):
        for ch in node.children:
            if isinstance(ch, NavigableString):
                out.append(str(ch))
            elif getattr(ch, 'name', None) == 'br':
                out.append('\n')
            elif getattr(ch, 'name', None) in _BLOCK_TAGS:
                out.append('\n')
                walk(ch)
                out.append('\n')
            else:
                walk(ch)
    walk(el)
    t = ''.join(out)
    t = _re.sub(r'[ \t\r\f\v]*\n[ \t\r\f\v]*', '\n', t)
    t = _re.sub(r'\n{3,}', '\n\n', t)
    return t.strip()


def init_db():
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.execute("PRAGMA journal_mode=WAL")
    conn.execute("PRAGMA busy_timeout=5000")
    return conn


def extract_list_page(page_num):
    """Fetch a list page and extract article URLs + titles + dates."""
    if page_num == 1:
        url = BASE_URL
    else:
        url = f"https://www.0735.com/toutiao/list-22.html?page={page_num}"

    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Failed to fetch page {page_num}: {e}", flush=True)
        return []

    if r.status_code != 200:
        print(f"  [ERROR] HTTP {r.status_code} for page {page_num}", flush=True)
        return []

    soup = BeautifulSoup(r.text, "html.parser")
    items = []

    for li in soup.select(".con h3"):
        a_tag = li.find("a")
        if not a_tag:
            continue

        title = a_tag.get_text(strip=True)
        href = a_tag.get("href", "").strip()
        if not href:
            continue
        full_url = urljoin(url, href)

        # Date from sibling .info2 .time
        parent_tag = li.parent
        info2 = parent_tag.find("div", class_="info2") if parent_tag else None
        time_span = info2.find("span", class_="time") if info2 else None
        date_str = time_span.get_text(strip=True) if time_span else ""
        # Skip relative dates like "26天前", "3分钟前"
        if re.search(r"[天时分钟秒]前", date_str):
            continue

        items.append({
            "title": title,
            "url": full_url,
            "date": date_str,
        })

    print(f"  Page {page_num}: {len(items)} items", flush=True)
    return items


def extract_detail(url):
    """Fetch a detail page and extract full content."""
    try:
        r = requests.get(url, headers=HEADERS, timeout=30)
        r.encoding = "utf-8"
    except Exception as e:
        print(f"  [ERROR] Failed to fetch {url}: {e}", flush=True)
        return None

    if r.status_code != 200:
        print(f"  [ERROR] HTTP {r.status_code} for {url}", flush=True)
        return None

    soup = BeautifulSoup(r.text, "html.parser")

    # Get date from the publish list
    pub_date = ""
    publish_ul = soup.select_one("ul.publish_l")
    if publish_ul:
        li = publish_ul.find("li")
        if li:
            pub_date = li.get_text(strip=True)

    # Get content from div.con#resizeIMG
    content_div = soup.select_one("div.con#resizeIMG")
    if not content_div:
        content_div = soup.select_one("div.con")
    if not content_div:
        content_div = soup.select_one("div.zx-content div.detail")

    content_text = ""
    attachments_list = []

    if content_div:
        # Process content - preserve paragraphs and extract attachments
        content_html = str(content_div)
        
        # Extract text from paragraphs
        paragraphs = []
        for p in content_div.find_all("p"):
            t = body_text(p)
            if t:
                paragraphs.append(t)
        content_text = "\n\n".join(paragraphs)

        # Extract file attachments (PDF, DOC, etc.)
        for a_tag in content_div.find_all("a", href=True):
            href = a_tag["href"]
            if re.search(r'\.(pdf|doc|docx|xls|xlsx|rar|zip)$', href, re.IGNORECASE):
                att_title = a_tag.get_text(strip=True) or href.split("/")[-1]
                full_url = urljoin(url, href)
                attachments_list.append({"name": att_title, "url": full_url})
        
        # Also look for Baidu Pan links (common on this site)
        for p in content_div.find_all("p"):
            text = p.get_text(strip=True)
            if "网盘" in text or "链接:" in text or "提取码" in text:
                attachments_list.append({"name": text[:80], "url": text})

    return {
        "content": content_text,
        "pub_date": pub_date,
        "attachments": json.dumps(attachments_list, ensure_ascii=False) if attachments_list else "",
    }


def crawl(test_mode=False, max_pages=MAX_PAGES):
    """Main crawl function."""
    print(f"[{SITE_NAME}] Starting crawl, max_pages={max_pages}", flush=True)
    conn = init_db()
    cur = conn.cursor()

    total_inserted = 0
    total_skipped = 0
    pages_crawled = 0

    for page_num in range(1, max_pages + 1):
        items = extract_list_page(page_num)
        if not items:
            print(f"  Page {page_num} empty, stopping", flush=True)
            break

        pages_crawled += 1
        for item in items:
            title = item["title"]
            url = item["url"]
            date = item["date"]

            # Check if already exists
            cur.execute("SELECT id FROM gov_raw WHERE page_url = ?", (url,))
            if cur.fetchone():
                total_skipped += 1
                continue

            # Fetch detail
            detail = extract_detail(url)
            if detail is None:
                total_skipped += 1
                continue

            content = detail["content"]

            # Handle empty content (PDF only pages)
            if len(content.strip()) < 20:
                content = f'<p><a href="{url}">{title}</a></p>\n（本文为PDF/网盘附件）'

            attachments = detail["attachments"]

            # Normalize date: "07-01" -> "2026-07-01" (current year)
            if date and re.match(r'^\d{2}-\d{2}$', date):
                date = f"2026-{date}"

            cur.execute(INSERT_SQL, (
                SITE_NAME, url, title, date or detail["pub_date"],
                content, CATEGORY, attachments,
            ))
            total_inserted += 1

            if test_mode and total_inserted >= 10:
                break

        conn.commit()

        if test_mode and total_inserted >= 10:
            print(f"  Test mode: stopping after {total_inserted} inserts", flush=True)
            break

        # Rate limit
        time.sleep(0.5)

    conn.close()
    print(f"[{SITE_NAME}] Done. Inserted={total_inserted}, Skipped={total_skipped}, Pages={pages_crawled}", flush=True)
    return total_inserted


if __name__ == "__main__":
    test_mode = "--test" in sys.argv
    mp = 5 if test_mode else MAX_PAGES
    
    # Support --json-output for local crawl
    json_output = False
    if "--json-output" in sys.argv:
        json_output = True
        test_mode = False  # Full crawl when using json
        
    if "--max-pages" in sys.argv:
        idx = sys.argv.index("--max-pages")
        if idx + 1 < len(sys.argv):
            mp = int(sys.argv[idx + 1])
    
    if json_output:
        # Run crawl, collect into JSON
        items_crawled = []
        total_inserted = 0
        
        for page_num in range(1, mp + 1):
            items = extract_list_page(page_num)
            if not items:
                break
            for item in items:
                detail = extract_detail(item["url"])
                if detail:
                    date = item["date"]
                    if date and re.match(r'^\d{2}-\d{2}$', date):
                        date = f"2026-{date}"
                    items_crawled.append({
                        "site_name": SITE_NAME,
                        "page_url": item["url"],
                        "title": item["title"],
                        "publish_date": date or detail["pub_date"],
                        "content": detail["content"],
                        "category": CATEGORY,
                        "attachments": detail["attachments"],
                    })
                    total_inserted += 1
                    if total_inserted % 20 == 0:
                        print(f"[progress] {total_inserted} crawled...", flush=True)
                time.sleep(0.3)
        
        # Write to output file
        out_path = "/tmp/0735_data.json"
        with open(out_path, "w", encoding="utf-8") as f:
            json.dump(items_crawled, f, ensure_ascii=False, indent=2)
        print(f"[done] {total_inserted} items saved to {out_path}", flush=True)
    else:
        crawl(test_mode=test_mode, max_pages=mp)
