#!/usr/bin/env python3
"""贵溪市人民政府-生态环境 (www.guixi.gov.cn)
URL: http://www.guixi.gov.cn/col/col6947/index.html?vc_xxgkarea=Y01000&number=G01000G01006G01003G01004G01004
CMS: Hanweb - XXGK信息公开系统
列表: POST /module/xxgk/search.jsp (infotypeId/colId/currpage)
详情: /art/.../art_6947_xxxxx.html -> div#zoom
"""

import sys, os, re, json, sqlite3
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
from datetime import datetime, timedelta

SEARCH_URL = "http://www.guixi.gov.cn/module/xxgk/search.jsp"
BASE_URL = "http://www.guixi.gov.cn"
COLUMN = "生态环境"
SITE = "贵溪市人民政府"
PROVINCE = "江西"
INFOTYPE_ID = "G10000G00006G00003G00000"
COL_ID = "6947"
PER_PAGE = 20
DB_PATH = "/root/search.db"

HEADERS = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
session = requests.Session()
session.headers.update(HEADERS)


def log(msg):
    sys.stderr.write(msg + "\n")
    sys.stderr.flush()


def fetch_list(page):
    """Fetch one page of list results."""
    data = {
        "divid": "div4",
        "infotypeId": INFOTYPE_ID,
        "jdid": "56",
        "area": "",
        "standardXxgk": "1",
        "colId": COL_ID,
        "currpage": str(page),
    }
    try:
        r = session.post(SEARCH_URL, data=data, timeout=20)
        r.encoding = "utf-8"
        return r.text
    except Exception as e:
        log(f"  [ERROR] list page {page}: {e}")
        return None


def parse_list(html):
    """Extract (title, url, date) from list page HTML."""
    items = []
    for m in re.finditer(
        r'<a title="([^\"]*?)"\s+target="_blank"\s+href="([^\"]+?)"[^>]*>.*?</a>\s*<b>\s*(\d{4}-\d{2}-\d{2})\s*</b>',
        html, re.DOTALL
    ):
        title = m.group(1).strip()
        url = m.group(2).strip()
        date_str = m.group(3).strip()
        if title and url:
            items.append({"title": title, "url": url, "date": date_str})
    return items


def get_total_pages(html):
    """Extract total number of pages."""
    m = re.search(r'共(\d+)条记录', html)
    if m:
        total = int(m.group(1))
        return (total + PER_PAGE - 1) // PER_PAGE
    return 0


def extract_detail(detail_url):
    """Extract content from detail page div#zoom."""
    try:
        resp = session.get(detail_url, timeout=15)
        resp.encoding = "utf-8"
    except Exception as e:
        log(f"  [ERROR] 详情页: {detail_url} - {e}")
        return "", "", "", []

    soup = BeautifulSoup(resp.text, "html.parser")

    # Title from meta
    title = ""
    meta_title = soup.select_one("meta[name=ArticleTitle]")
    if meta_title and meta_title.get("content"):
        title = meta_title["content"].strip()

    # Date from meta
    date_str = ""
    meta_date = soup.select_one("meta[name=pubdate]")
    if meta_date and meta_date.get("content"):
        date_str = meta_date["content"].strip()[:10]

    # Content from div#zoom with paragraph/table/attachment fixes
    content_div = soup.select_one("div#zoom")
    content, attachments = extract_content(content_div, title, detail_url)

    return title, date_str, content, attachments


MARKER = "\x00P\x00"

# Attachment file extensions
ATTACH_EXTS = ('.doc', '.docx', '.pdf', '.xls', '.xlsx', '.ppt', '.pptx',
               '.zip', '.rar', '.7z', '.tar', '.gz', '.txt',
               'downfile.jsp', 'download.jsp', 'downfile', 'download')


def _is_attachment(href):
    """Check if a URL points to an attachment file."""
    if not href:
        return False
    low = href.lower()
    return any(ext in low for ext in ATTACH_EXTS)


def extract_content(content_div, title, detail_url):
    """Extract formatted content with proper paragraphs, table preservation, and attachment links.

    Returns: (content_text, attachments_list)
    """
    if not content_div:
        if title:
            return f"[{title}]({detail_url})", []
        return "", []

    attachments = []

    # 0) Extract attachment links to markdown BEFORE any stripping
    for a_tag in content_div.find_all("a", href=True):
        href = a_tag["href"]
        text = a_tag.get_text(strip=True)
        if _is_attachment(href):
            # Resolve relative URLs
            full_url = urljoin(BASE_URL, href)
            # Replace the <a> tag with its markdown equivalent
            md_link = f"[{text}]({full_url})" if text else f"[附件]({full_url})"
            a_tag.replace_with(md_link)
            attachments.append({"name": text or "附件", "url": full_url})
        elif href.startswith("http") and not href.startswith("javascript"):
            # External link - keep as markdown
            full_url = href
            md_link = f"[{text}]({full_url})" if text else ""
            if md_link and text:
                a_tag.replace_with(md_link)

    # 1) Save and remove tables
    tables_html = []
    for table in content_div.find_all("table"):
        try:
            tables_html.append(str(table))
        except Exception:
            tables_html.append(table.get_text(separator=" ", strip=True))
        table.decompose()

    # 2) Insert paragraph markers for <p>, headings, and <br>
    for tag in content_div.find_all(["p", "h1", "h2", "h3", "h4", "h5", "h6"]):
        tag.insert(0, MARKER)
        tag.append(MARKER)
    for br in content_div.find_all("br"):
        br.replace_with(MARKER)

    # 3) Unwrap inline tags with empty separator (preserves already-converted markdown)
    for tag in content_div.find_all(["span", "b", "strong", "font", "em", "i", "u", "s"]):
        tag.unwrap()

    # 4) Extract text - markdown links survive get_text() as text
    text = content_div.get_text(separator="", strip=True)

    # 5) Clean up markers
    text = re.sub(r"\x00P\x00(\s*\x00P\x00)+", "\x00P\x00", text)
    text = text.replace("\x00P\x00", "\n\n")
    text = re.sub(r"\n{3,}", "\n\n", text)
    text = text.strip()

    # 6) Append tables at end
    for tbl_html in tables_html:
        text += f"\n\n{tbl_html}"

    # 7) Empty content fallback
    if not text.strip() or len(text.strip()) < 20:
        if title:
            text = f"[{title}]({detail_url})"

    return text, attachments


def crawl_all(months_back=36):
    """Full crawl all pages."""
    cutoff = datetime.now() - timedelta(days=months_back * 30) if months_back else None

    html = fetch_list(1)
    if not html:
        log("[ERROR] 无法获取第一页")
        return
    total_pages = get_total_pages(html)
    log(f"共 {total_pages} 页")

    all_items = []
    seen_urls = set()

    for page in range(1, total_pages + 1):
        if page > 1:
            html = fetch_list(page)
            if not html:
                continue
        items = parse_list(html)
        new_count = 0
        for item in items:
            if item["url"] not in seen_urls:
                if cutoff and item["date"]:
                    try:
                        d = datetime.strptime(item["date"], "%Y-%m-%d")
                        if d < cutoff:
                            continue
                    except ValueError:
                        pass
                seen_urls.add(item["url"])
                all_items.append(item)
                new_count += 1
        log(f"  [PAGE {page}/{total_pages}] +{new_count} 条")
        if new_count == 0 and page > 1:
            log("  [STOP] no new items")
            break

    log(f"\n共 {len(all_items)} 条待爬详情")
    for idx, item in enumerate(all_items, 1):
        title, date_str, content, attachments = extract_detail(item["url"])
        if not item["date"] and date_str:
            item["date"] = date_str
        if title:
            item["title"] = title
        item["content"] = content
        item["attachments"] = attachments
        output_item(item, idx)
        if idx % 10 == 0:
            log(f"  [PROGRESS] {idx}/{len(all_items)}")

    added, skipped = push_to_db(all_items)
    log(f"[DB] 入库: 新增{added} 跳过{skipped}")
    log(f"\n[DONE] 共爬取 {len(all_items)} 条")


def crawl_incremental():
    """Incremental: only first page."""
    html = fetch_list(1)
    if not html:
        return
    items = parse_list(html)
    log(f"[LIST] page 1 -> {len(items)} 条")
    for idx, item in enumerate(items, 1):
        title, date_str, content, attachments = extract_detail(item["url"])
        if not item["date"] and date_str:
            item["date"] = date_str
        if title:
            item["title"] = title
        item["content"] = content
        item["attachments"] = attachments
        output_item(item, idx)
    added, skipped = push_to_db(items)
    log(f"[DB] 入库: 新增{added} 跳过{skipped}")
    log(f"[DONE] 增量爬取 {len(items)} 条")


def output_item(item, idx):
    record = {
        "title": item.get("title", ""),
        "page_url": item.get("url", ""),
        "publish_date": item.get("date", ""),
        "content": item.get("content", ""),
        "attachments": item.get("attachments", []),
        "site_name": f"{SITE}-{COLUMN}",
        "column": COLUMN,
        "province": PROVINCE,
    }
    print(json.dumps(record, ensure_ascii=False))


def push_to_db(items):
    """INSERT OR IGNORE 直插 gov_raw; FTS 触发器 trg_gov_raw_fts_ins 自动同步 gov_search"""
    if not items:
        return 0, 0
    if not os.path.exists(DB_PATH):
        log(f"  未发现 {DB_PATH}, 跳过入库")
        return 0, 0
    conn = sqlite3.connect(DB_PATH, timeout=290)
    conn.execute("PRAGMA busy_timeout=290000")
    added = skipped = 0
    cur = conn.cursor()
    for it in items:
        content = it.get("content", "") or ""
        attachments = it.get("attachments", [])
        att_json = json.dumps(attachments, ensure_ascii=False) if attachments else ""
        url = it.get("url", "")
        try:
            cur.execute(
                """INSERT OR IGNORE INTO gov_raw
                   (site_name, source_url, page_url, title, publish_date, content, summary, category, script_name, attachments, group_name, industry)
                   VALUES (?,?,?,?,?,?,?,?,?,?,?,?)""",
                (f"{SITE}-{COLUMN}", url, url, it.get("title", ""), it.get("date", ""),
                 content, content[:500], COLUMN, "crawl_guixi_sthj_xxgk.py", att_json,
                 PROVINCE, "other")
            )
            if cur.rowcount > 0:
                added += 1
            else:
                skipped += 1
        except sqlite3.Error as e:
            log(f"  DB err: {e}")
            skipped += 1
        conn.commit()
    conn.close()
    return added, skipped


if __name__ == "__main__":
    mode = sys.argv[1] if len(sys.argv) > 1 else "incremental"
    if mode == "incremental":
        crawl_incremental()
    elif mode == "full":
        months = int(sys.argv[2]) if len(sys.argv) > 2 else 36
        crawl_all(months)
    elif mode == "list":
        html = fetch_list(1)
        items = parse_list(html)
        log(f"共 {len(items)} 条")
        for it in items[:5]:
            log(f"  {it['date']} | {it['title'][:40]} | {it['url']}")
    else:
        log(f"Usage: {sys.argv[0]} [incremental|full [months]]")
