#!/usr/bin/env python3
"""
滦州市人民政府-服务事项 爬虫
站点: http://www.luanxian.gov.cn/index.php?m=content&c=index&a=lists&catid=854
CMS: 易通PHP CMS（信息公开平台模板）
"""
import re, sys, json, time, urllib.request, urllib.error, urllib.parse, sqlite3
from bs4 import BeautifulSoup

BASE_URL = "http://www.luanxian.gov.cn"
LIST_URL = BASE_URL + "/index.php?m=content&c=index&a=lists&catid=854"
SITE_NAME = "滦州市人民政府-服务事项"
DB_PATH = "/mnt/data/search.db"
DELAY = 1.5

def fetch(url):
    try:
        req = urllib.request.Request(url, headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
        })
        resp = urllib.request.urlopen(req, timeout=15)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        print("  [FETCH ERROR] %s: %s" % (url, e), flush=True)
        return None

def parse_list(html):
    """解析列表页: div > div > a + span(日期)"""
    items = []
    # 找列表区域
    for m in re.finditer(
        r'<a[^>]*href="([^"]*show&catid=854&id=(\d+))"[^>]*style="[^"]*">([^<]+)</a>',
        html, re.DOTALL
    ):
        url = m.group(1).strip()
        art_id = m.group(2)
        title = m.group(3).strip()
        if not url.startswith("http"):
            url = urllib.parse.urljoin(BASE_URL, url)
        # 找同一id的日期链接
        date_m = re.search(r'<a[^>]*href="[^"]*show&catid=854&id=' + art_id + r'"[^>]*>(\d{4}-\d{2}-\d{2})</a>', html)
        date = date_m.group(1) if date_m else ""
        if title:
            items.append((title, url, date))
    return items

def parse_detail(html, url):
    """提取正文: #conN.container"""
    attachments = []

    # 正文容器: <div id="conN">
    m = re.search(r'<div[^>]*id="conN"[^>]*>(.*?)</div>\s*<div[^>]*class="clearfloat"', html, re.DOTALL)
    if not m:
        m = re.search(r'<div[^>]*id="conN"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not m:
        for pat in [r'<div[^>]*class="ftext"[^>]*>(.*?)</div>',
                    r'<div[^>]*class="content"[^>]*>(.*?)</div>',
                    r'<div[^>]*id="zoom"[^>]*>(.*?)</div>',
                    r'<div[^>]*class="article"[^>]*>(.*?)</div>',
                    r'<div[^>]*class="conTxt"[^>]*>(.*?)</div>']:
            m = re.search(pat, html, re.DOTALL)
            if m:
                break
    if not m:
        return "", []

    container = m.group(1)
    # 去掉内嵌样式
    container = re.sub(r'<style[^>]*>.*?</style>', '', container, flags=re.DOTALL)
    container = re.sub(r'<script[^>]*>.*?</script>', '', container, flags=re.DOTALL)

    # 表格位置
    table_ranges = []
    for tm in re.finditer(r"<table[^>]*>.*?</table>", container, re.DOTALL):
        table_ranges.append((tm.start(), tm.end()))

    def inside_table(pos):
        return any(ts <= pos <= te for ts, te in table_ranges)

    def has_block_children(el_html):
        body = re.sub(r'^<p[^>]*>', '', el_html)
        body = re.sub(r'</p>\s*$', '', body)
        return bool(re.search(r'<(div|p|table|ul|ol|h[1-6])\b', body))

    def html_table_to_markdown(html_table):
        # 保留 HTML 表格结构（不再转 markdown 管道表）
        return html_table
    elements = []
    for m in re.finditer(r"<(p|table|img)\b[^>]*>", container, re.DOTALL):
        start, m_end, tag = m.start(), m.end(), m.group(1)
        if tag in ("p", "table"):
            close_tag = "</" + tag + ">"
            end_pos = container.find(close_tag, m_end)
            if end_pos >= 0:
                el_html = container[start:end_pos + len(close_tag)]
            else:
                ns = container.find("<", m_end)
                el_html = container[start:ns] if ns >= 0 else container[start:]
        elif tag == "img":
            cp = container.find(">", m_end)
            if cp >= 0:
                el_html = container[start:cp + 1]
            else:
                continue
        else:
            continue

        if tag == "p":
            if inside_table(start):
                continue
            if has_block_children(el_html):
                continue
            text = re.sub(r"<[^>]+>", "", el_html).strip()
            text = re.sub(r"\s+", " ", text)
            if text:
                elements.append((start, text))
        elif tag == "table":
            if el_html.count("<td") > 0:
                md = html_table_to_markdown(el_html)
                if md:
                    elements.append((start, md))
        elif tag == "img":
            src_m = re.search(r'src="([^"]+)"', el_html)
            if src_m:
                src = urllib.parse.urljoin(url, src_m.group(1))
                elements.append((start, "![](%s)" % src))

    elements.sort(key=lambda x: x[0])
    content = "\n\n".join(text for _, text in elements)

    # 附件（全文查找，不限于#conN）
    for a_href, a_text in re.findall(
        r'<a[^>]*href="([^"]+\.(?:doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar|7z|txt|wps|et|DOC))"[^>]*>([^<]+)</a>',
        html, re.DOTALL
    ):
        attachments.append({"title": a_text.strip(), "url": urllib.parse.urljoin(url, a_href)})

    # 正文为空但有附件时，加PDF占位提示
    if (not content or not content.strip()) and attachments:
        content = "【该文档内容详见附件PDF/文档】\n"
        for att in attachments:
            content += "附件: " + att["title"] + " - " + att["url"] + "\n"

    return content, attachments

def get_max_page(html):
    """从列表页取最大页码"""
    pages = re.findall(r'catid=854&page=(\d+)', html)
    if pages:
        return max(int(p) for p in pages)
    return 1

def crawl(max_pages=5):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    cur.execute("DELETE FROM gov_search WHERE site_name=?", (SITE_NAME,))
    cur.execute("DELETE FROM gov_raw WHERE site_name=?", (SITE_NAME,))
    conn.commit()
    print("[清理] 已清理旧数据", flush=True)

    new_count = skip_count = 0

    print("[分析] 获取列表第1页...", flush=True)
    time.sleep(DELAY)
    html = fetch(LIST_URL)
    if not html:
        print("[失败] 无法获取列表", flush=True)
        conn.close()
        return
    total_pages = get_max_page(html)
    pages_to_crawl = min(max_pages, total_pages) if max_pages > 0 else total_pages
    print("[分析] 共%d页，本次爬取%d页" % (total_pages, pages_to_crawl), flush=True)

    # 第1页
    items = parse_list(html)
    if items:
        print("  第1页: 找到 %d 条" % len(items), flush=True)
        for title, detail_url, date in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
            if cur.fetchone():
                skip_count += 1
                continue
            time.sleep(DELAY)
            detail_html = fetch(detail_url)
            if not detail_html:
                skip_count += 1
                continue
            content, attachments = parse_detail(detail_html, detail_url)
            aj = json.dumps(attachments, ensure_ascii=False)
            summary = content[:200] if content else title
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name,page_url,title,content,publish_date,attachments,summary) VALUES(?,?,?,?,?,?,?)",
                (SITE_NAME, detail_url, title, content, date, aj, summary)
            )
            if cur.rowcount > 0:
                new_count += 1
                status = "[正文]" if content else "[无文本]"
                print("  %s %s -> %d chars, %d附件" %
                      (status, title[:35], len(content), len(attachments)), flush=True)
        conn.commit()

    # 第2页+
    for page in range(2, pages_to_crawl + 1):
        page_url = LIST_URL + "&page=%d" % page
        print("\n[分页] 第%d页: %s" % (page, page_url), flush=True)
        time.sleep(DELAY)
        html = fetch(page_url)
        if not html:
            print("  [完成] 获取失败", flush=True)
            break
        items = parse_list(html)
        if not items:
            print("  [完成] 无数据", flush=True)
            break
        print("  找到 %d 条" % len(items), flush=True)
        for title, detail_url, date in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
            if cur.fetchone():
                skip_count += 1
                continue
            time.sleep(DELAY)
            detail_html = fetch(detail_url)
            if not detail_html:
                skip_count += 1
                continue
            content, attachments = parse_detail(detail_html, detail_url)
            aj = json.dumps(attachments, ensure_ascii=False)
            summary = content[:200] if content else title
            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name,page_url,title,content,publish_date,attachments,summary) VALUES(?,?,?,?,?,?,?)",
                (SITE_NAME, detail_url, title, content, date, aj, summary)
            )
            if cur.rowcount > 0:
                new_count += 1
                status = "[正文]" if content else "[无文本]"
                print("  %s %s -> %d chars, %d附件" %
                      (status, title[:35], len(content), len(attachments)), flush=True)
        conn.commit()

    conn.close()
    print("\n[DONE] %s: 新增=%d, 跳过=%d" % (SITE_NAME, new_count, skip_count), flush=True)

if __name__ == "__main__":
    _pages_args = [int(a.split('=', 1)[1]) for a in sys.argv if a.startswith('--pages=')]
    _pages_args = _pages_args or [int(a) for a in sys.argv[1:] if a.isdigit()]
    pages = _pages_args[0] if _pages_args else 5
    crawl(pages)
