#!/usr/bin/env python3
"""
陆良县人民政府-部门动态 爬虫
站点: https://www.luliang.gov.cn/news/bmdt.html
CMS: 自定义
"""
import re, sys, json, time, urllib.request, urllib.error, urllib.parse, sqlite3
from bs4 import BeautifulSoup

BASE_URL = "https://www.luliang.gov.cn"
SITE_NAME = "陆良县人民政府-部门动态"
DB_PATH = "/mnt/data/search.db"
DELAY = 1.5

def fetch(url):
    try:
        req = urllib.request.Request(url, headers={
            "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"
        })
        resp = urllib.request.urlopen(req, timeout=15)
        return resp.read().decode("utf-8", errors="replace")
    except Exception as e:
        print("  [FETCH ERROR] %s: %s" % (url, e), flush=True)
        return None

def parse_list(html):
    """解析列表页: <ol class=\"news_list\"> -> <li> -> <a>标题</a>日期"""
    items = []
    for m in re.finditer(
        r'<a[^>]*href="(https?://[^"]*?/news/bmdt/\d+\.html)"[^>]*>(.*?)</a>(\d{4}-\d{2}-\d{2})',
        html, re.DOTALL
    ):
        url = m.group(1).strip()
        title = re.sub(r'<[^>]+>', '', m.group(2)).strip()
        title = re.sub(r'\s+', ' ', title)
        date = m.group(3).strip()
        if not url.startswith("http"):
            url = urllib.parse.urljoin(BASE_URL, url)
        items.append((title, url, date))
    return items

def parse_detail(html, url):
    """提取正文: div.web_con -> p, img, table"""
    attachments = []

    # 正文容器: <div class="web_con">
    m = re.search(r'<div[^>]*class="web_con[^"]*"[^>]*>(.*?)</div>\s*</div>', html, re.DOTALL)
    if not m:
        m = re.search(r'<div[^>]*class="web_con[^"]*"[^>]*>(.*?)</div>', html, re.DOTALL)
    if not m:
        return "", []

    container = m.group(1)

    # 表格位置
    table_ranges = []
    for tm in re.finditer(r"<table[^>]*>.*?</table>", container, re.DOTALL):
        table_ranges.append((tm.start(), tm.end()))

    def inside_table(pos):
        return any(ts <= pos <= te for ts, te in table_ranges)

    def has_block_children(el_html):
        body = re.sub(r'^<p[^>]*>', '', el_html)
        body = re.sub(r'</p>\s*$', '', body)
        return bool(re.search(r'<(div|p|table|ul|ol|h[1-6])\b', body))

    def html_table_to_markdown(html_table):
        # 保留 HTML 表格结构（不再转 markdown 管道表）
        return html_table
    elements = []
    for m in re.finditer(r"<(p|table|img)\b[^>]*>", container, re.DOTALL):
        start, m_end, tag = m.start(), m.end(), m.group(1)
        if tag in ("p", "table"):
            close_tag = "</" + tag + ">"
            end_pos = container.find(close_tag, m_end)
            if end_pos >= 0:
                el_html = container[start:end_pos + len(close_tag)]
            else:
                ns = container.find("<", m_end)
                el_html = container[start:ns] if ns >= 0 else container[start:]
        elif tag == "img":
            cp = container.find(">", m_end)
            if cp >= 0:
                el_html = container[start:cp + 1]
            else:
                continue
        else:
            continue

        if tag == "p":
            if inside_table(start):
                continue
            if has_block_children(el_html):
                continue
            text = re.sub(r"<[^>]+>", "", el_html).strip()
            text = re.sub(r"\s+", " ", text)
            if text:
                elements.append((start, text))
        elif tag == "table":
            if el_html.count("<td") > 0:
                md = html_table_to_markdown(el_html)
                if md:
                    elements.append((start, md))
        elif tag == "img":
            src_m = re.search(r'src="([^"]+)"', el_html)
            if src_m:
                src = urllib.parse.urljoin(url, src_m.group(1))
                elements.append((start, "![](%s)" % src))

    elements.sort(key=lambda x: x[0])
    content = "\n\n".join(text for _, text in elements)

    # 附件
    for a_href, a_text in re.findall(
        r'<a[^>]*href="([^"]+\.(?:doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar|7z|txt|wps|et))"[^>]*>([^<]+)</a>',
        html, re.DOTALL
    ):
        attachments.append({"title": a_text.strip(), "url": urllib.parse.urljoin(url, a_href)})

    return content, attachments

def crawl(max_pages=5):
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    new_count = skip_count = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = BASE_URL + "/news/bmdt.html"
        else:
            list_url = BASE_URL + "/news/bmdt.html?page=%d" % page

        print("[分页] 第%d页: %s" % (page, list_url), flush=True)
        time.sleep(DELAY)
        html = fetch(list_url)
        if not html:
            print("  [完成] 获取失败", flush=True)
            break

        items = parse_list(html)
        if not items:
            print("  [完成] 无数据", flush=True)
            break

        print("  找到 %d 条" % len(items), flush=True)

        for title, detail_url, date in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
            if cur.fetchone():
                skip_count += 1
                continue

            time.sleep(DELAY)
            detail_html = fetch(detail_url)
            if not detail_html:
                skip_count += 1
                continue

            content, attachments = parse_detail(detail_html, detail_url)
            aj = json.dumps(attachments, ensure_ascii=False)
            summary = content[:200] if content else title

            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name,page_url,title,content,publish_date,attachments,summary) VALUES(?,?,?,?,?,?,?)",
                (SITE_NAME, detail_url, title, content, date, aj, summary)
            )
            if cur.rowcount > 0:
                new_count += 1
                if content:
                    print("  [正文] %s -> %d chars, %d附件" %
                          (title[:35], len(content), len(attachments)), flush=True)
                else:
                    print("  [无文本] %s" % title[:35], flush=True)

        conn.commit()

    conn.close()
    print("\n[DONE] %s: 新增=%d, 跳过=%d" % (SITE_NAME, new_count, skip_count), flush=True)

if __name__ == "__main__":
    _pages_args = [int(a.split('=', 1)[1]) for a in sys.argv if a.startswith('--pages=')]
    _pages_args = _pages_args or [int(a) for a in sys.argv[1:] if a.isdigit()]
    pages = _pages_args[0] if _pages_args else 5
    crawl(pages)
