#!/usr/bin/env python3
"""衡阳松木经济开发区 - 通知公告 (www.hengyang.gov.cn/hysm/zwgk/tzgg)
CMS: 博删DFS模板系统
列表: index.html + pages/{N}.html (N=2~26, 共26页)
详情: title=div.zjym_j, date=div.zjym_x, content=div.zjym_d (保留HTML)

用法:
  python3 crawl_hengyang_hysm.py          # 全量26页
  python3 crawl_hengyang_hysm.py 1        # 增量1页
"""
import re, requests, sqlite3, os, sys, time
from datetime import datetime

BASE = "https://www.hengyang.gov.cn"
LIST_DIR = "/hysm/zwgk/tzgg"
SITE_NAME = "衡阳松木经济开发区-通知公告"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/125.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9,en;q=0.8",
}
TOTAL_PAGES_ALL = 26


def get_conn():
    conn = sqlite3.connect(DB_PATH, timeout=60)
    conn.row_factory = sqlite3.Row
    return conn


def http_get(url):
    try:
        resp = requests.get(url, headers=HEADERS, timeout=30)
        resp.encoding = "utf-8"
        if resp.status_code == 200 and len(resp.text) > 500:
            return resp.text
        return None
    except Exception as e:
        return None


def get_list_url(page):
    if page == 1:
        return f"{BASE}{LIST_DIR}/index.html"
    return f"{BASE}{LIST_DIR}/pages/{page}.html"


def parse_list(html):
    items = []
    for m in re.finditer(
        r'<li[^>]*>\s*<a[^>]*href="([^"]+)"[^>]*>(.*?)</a>\s*<span>\s*\[?(\d{4}-\d{2}-\d{2})\]?\s*</span>',
        html, re.DOTALL
    ):
        href = m.group(1).strip()
        a_tag = m.group(0)
        title_attr = re.search(r'title="([^"]+)"', a_tag)
        title = title_attr.group(1) if title_attr else m.group(2).strip()
        date = m.group(3).strip()
        if len(title) < 8 and title in ['首页', '通知公告', '政务公开', '经开区介绍']:
            continue
        if not href.startswith("http"):
            href = BASE + href if href.startswith("/") else f"{BASE}{LIST_DIR}/{href}"
        items.append({"title": title.strip(), "url": href, "date": date})
    return items


def extract_detail(url, title_from_list):
    html = http_get(url)
    if not html:
        return None

    title = title_from_list
    tm = re.search(r'<div[^>]*class="zjym_j"[^>]*>(.*?)</div>', html, re.DOTALL)
    if tm:
        t = re.sub(r'<[^>]+>', '', tm.group(1)).strip()
        if t:
            title = t

    date = ""
    dm = re.search(r'发布日期[：:]\s*(\d{4}-\d{2}-\d{2})', html)
    if dm:
        date = dm.group(1)

    content = ""
    cm = re.search(r'<div[^>]*class="zjym_d"[^>]*>(.*?)</div>', html, re.DOTALL)
    if cm:
        raw = cm.group(1)
        raw = re.sub(r'^\s*<ul[^>]*>\s*<li[^>]*>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'</li>\s*</ul>\s*$', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<script[^>]*>.*?</script>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<link[^>]*>', '', raw, flags=re.DOTALL)
        raw = re.sub(r'<style[^>]*>.*?</style>', '', raw, flags=re.DOTALL)
        raw = raw.strip()
        if raw:
            content = raw

    # 附件
    attachments = []
    pdf_links = re.findall(r'data-pdf="([^"]+)"', html)
    pdf_names = re.findall(r'<span[^>]*class="edui-pdf"[^>]*>(.*?)</span>', html)
    for i, pdf_url in enumerate(pdf_links):
        name = pdf_names[i] if i < len(pdf_names) else f"PDF附件{i+1}"
        full_url = BASE + pdf_url if pdf_url.startswith("/") else pdf_url
        attachments.append({"name": name.strip(), "url": full_url})
    for am in re.finditer(
        r'<a[^>]*href="([^"]*\.(pdf|doc|docx|xls|xlsx|zip|rar))"[^>]*>([^<]+)</a>',
        html, re.I
    ):
        url2 = am.group(1)
        name2 = am.group(3).strip()
        if url2.startswith("/"):
            url2 = BASE + url2
        attachments.append({"name": name2, "url": url2})

    # PDF空内容处理：正文<20字符时嵌入原文+PDF链接
    summary = re.sub(r'<[^>]+>', ' ', content).strip()
    summary = re.sub(r'\s+', ' ', summary)[:500]
    if not summary:
        summary = title[:300]

    text_len = len(re.sub(r'<[^>]+>', '', content).strip())
    if text_len < 20 and attachments:
        # 空内容，嵌入原文链接
        content = f'<p><a href="{url}">{title}</a></p>' + content
        for att in attachments:
            content += f'\n<p><a href="{att["url"]}">{att["name"]}</a></p>'
    else:
        for att in attachments:
            if att["url"] not in (content or ""):
                content += f'\n<p><a href="{att["url"]}">{att["name"]}</a></p>'

    return {
        "title": title,
        "url": url,
        "date": date,
        "content": content,
        "summary": summary,
        "attachments": attachments,
    }


def main():
    # Parse args
    incremental = False
    pages_override = None
    args_list = list(sys.argv[1:])
    i = 0
    while i < len(args_list):
        arg = args_list[i]
        if arg in ("1", "incremental", "--incremental"):
            incremental = True
        elif arg == "--pages" and i + 1 < len(args_list):
            pages_override = int(args_list[i + 1])
            i += 1
        i += 1

    if pages_override:
        total_pages = pages_override
    else:
        total_pages = 1 if incremental else TOTAL_PAGES_ALL
    mode = "增量" if incremental else "全量"

    print(f"\n{'='*50}", flush=True)
    print(f"🏠 {SITE_NAME} ({mode})", flush=True)
    print(f"   DB: {DB_PATH}", flush=True)
    print(f"{'='*50}", flush=True)

    new_count = 0
    skip_count = 0
    error_count = 0

    for page in range(1, total_pages + 1):
        url = get_list_url(page)
        print(f"\n📄 列表页 {page}/{total_pages}: {url}", flush=True)
        html = http_get(url)
        if not html:
            print(f"  [WARN] 获取失败，跳过", flush=True)
            continue

        items = parse_list(html)
        if not items:
            print(f"  [WARN] 无文章链接，结束分页", flush=True)
            break

        print(f"  发现 {len(items)} 篇文章", flush=True)

        for i, item in enumerate(items):
            detail = extract_detail(item["url"], item["title"])
            if not detail:
                print(f"  [{i+1}] ✗ {item['title'][:40]}... (详情获取失败)", flush=True)
                error_count += 1
                continue

            conn = get_conn()
            try:
                content = (detail["content"] or "")[:500000]
                summary = (detail["summary"] or "")[:300]
                conn.execute(
                    "INSERT OR IGNORE INTO gov_raw (title, page_url, content, publish_date, summary, site_name, tags) "
                    "VALUES (?, ?, ?, ?, ?, ?, ?)",
                    (detail["title"], detail["url"], content, detail["date"] or "", summary, SITE_NAME, "环评阶段")
                )
                changed = conn.total_changes > 0
                conn.commit()
                if changed:
                    new_count += 1
                    if incremental:
                        print(f"  [{i+1}] ✅ {detail['title'][:50]} ({detail['date']})", flush=True)
                else:
                    skip_count += 1
            except Exception as e:
                if incremental:
                    print(f"  [ERROR] {e}", flush=True)
                error_count += 1
            finally:
                conn.close()

            time.sleep(0.3)

    print(f"\n{'─'*30}", flush=True)
    print(f"✅ 新增: {new_count} | 跳过: {skip_count} | 错误: {error_count}", flush=True)

    if new_count > 0:
        conn = get_conn()
        try:
            conn.execute(
                "INSERT OR IGNORE INTO gov_search(rowid, title, site_name, summary, rowid_col) "
                "SELECT id, title, site_name, substr(content, 1, 500), id "
                "FROM gov_raw WHERE site_name=?", (SITE_NAME,)
            )
            conn.commit()

        except Exception as e:
            print(f"  [WARN] FTS同步:{e}", flush=True)
        finally:
            conn.close()

    print(f"📤 完成！共 {new_count} 条新数据", flush=True)


if __name__ == "__main__":
    main()
