#!/usr/bin/env python3
"""
泸州市人民政府-公示公告 爬虫 (v2)
CMS: PowerCMS
列表: /xw/gsgg (page1), /xw/gsgg_N (page2+)
详情: /xw/gsgg/{zfgs|zfgg}/content_NNNNNN
Fix v2: 清理"标题："前缀, 修复表格内容重复(跳过<td>内的<p>)
"""
import os, re, sys, json, time
from bs4 import BeautifulSoup
from urllib import request
from urllib.parse import urljoin
import ssl
import urllib.parse

SITE_NAME = "泸州市人民政府-公示公告"
BASE_URL = "https://www.luzhou.gov.cn"
LIST_BASE = "/xw/gsgg"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
DELAY = 5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

ssl._create_default_https_context = ssl._create_unverified_context


def fetch(url, timeout=30):
    req = request.Request(url, headers=HEADERS)
    try:
        resp = request.urlopen(req, timeout=timeout)
        return resp.read().decode("utf-8", errors="ignore")
    except Exception as e:
        print("  [ERROR] %s: %s" % (url, e), file=sys.stderr)
        return None


def parse_list(html):
    """从列表页提取文章信息. ul.newsList > li > span.date + a"""
    items = []
    for m in re.finditer(r'<span\s+class="date">(\d{4}-\d{2}-\d{2})</span>\s*<a\s+href="([^"]+)"[^>]*title="([^"]*)"', html):
        date = m.group(1)
        href = m.group(2)
        title = m.group(3).strip()
        # 清理"标题："前缀
        title = re.sub(r"^标题：\s*", "", title)
        if not title or len(title) < 5:
            continue
        full_url = href if href.startswith("http") else (BASE_URL + href)
        items.append((title, full_url, date))
    return items


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_detail(html, url):
    """
    提取正文(含表格、图片)和附件.
    修复: 跳过<td>内部的<p>防重复, 识别多种容器.
    """
    content_parts = []
    attachments = []

    # 1. 找正文容器
    container = None
    for pat in [
        r'<div[^>]*class="conTxt"[^>]*>(.*?)</div>\s*</div>',
        r'<div[^>]*class="conTxt"[^>]*>(.*?)</div>',
        r'<div[^>]*class="content"[^>]*>(.*?)</div>\s*</div>',
        r'<div[^>]*class="article"[^>]*>(.*?)</div>\s*</div>',
        r'<div[^>]*id="zoom"[^>]*>(.*?)</div>',
    ]:
        m = re.search(pat, html, re.DOTALL)
        if m:
            container = m.group(1)
            break

    if not container:
        body_m = re.search(r"<body[^>]*>(.*?)</body>", html, re.DOTALL)
        if body_m:
            for p in re.findall(r"<p[^>]*>(.*?)</p>", body_m.group(1), re.DOTALL):
                text = re.sub(r"<[^>]+>", "", p).strip()
                if text:
                    content_parts.append(text)
        content = "\n\n".join(content_parts)
    else:
        # 先找所有table的位置区间
        table_ranges = []
        for tm in re.finditer(r"<table[^>]*>.*?</table>", container, re.DOTALL):
            table_ranges.append((tm.start(), tm.end()))

        def is_inside_table(pos):
            for ts, te in table_ranges:
                if ts <= pos <= te:
                    return True
            return False

        # 提取所有p, table, img并排序
        elements = []

        # 处理table (先提取并移除table_ranges内p)
        for pos, tag, el_html in _extract_elements_sorted(container):
            if tag == "table":
                td_count = el_html.count("<td")
                if td_count > 0:
                    md = html_table_to_html(el_html)
                    if md:
                        elements.append((pos, md))
            elif tag == "img":
                src_m = re.search(r'src="([^"]+)"', el_html)
                alt_m = re.search(r'alt="([^"]*)"', el_html)
                if src_m:
                    src = src_m.group(1)
                    # 跳过装饰性图标
                    if "icon_" in src or "/filetypeimages/" in src:
                        continue
                    alt = alt_m.group(1) if alt_m else ""
                    full_src = urljoin(url, src)
                    elements.append((pos, "![%s](%s)" % (alt, full_src)))
            elif tag == "p":
                # 跳过在table内部的p (防止表格内容重复)
                if is_inside_table(pos):
                    continue
                # 跳过包含table的p (某些CMS把table包在p里)
                if "<table" in el_html or "<td" in el_html or "<th" in el_html:
                    continue
                text = re.sub(r"<[^>]+>", "", el_html).strip()
                text = text.replace("&nbsp;", " ")
                text = re.sub(r"\s+", " ", text)
                if text.strip():
                    elements.append((pos, text))

        # 按位置排序
        elements.sort(key=lambda x: x[0])
        content = "\n\n".join(text for _, text in elements)

    # 附件提取
    file_exts = r"\.(?:doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar|7z|txt|wps|et)"
    for a_href, a_text in re.findall(
        r'<a[^>]*href="([^"]+' + file_exts + r')"[^>]*>([^<]+)</a>',
        html, re.DOTALL
    ):
        full_url = urljoin(url, a_href)
        attachments.append({"title": a_text.strip(), "url": full_url})

    return content, attachments


def _extract_elements_sorted(container):
    """从容器中提取所有p, table, img元素."""
    results = []
    # 找所有起始标签
    tag_starts = []
    for m in re.finditer(r'<(p|table|img)\b[^>]*>', container, re.DOTALL):
        tag_starts.append((m.start(), m.end(), m.group(1)))

    for start, m_end, tag in tag_starts:
        if tag in ("p", "table"):
            # 找结束标签
            close_tag = "</" + tag + ">"
            end_pos = container.find(close_tag, m_end)
            if end_pos >= 0:
                el_html = container[start:end_pos + len(close_tag)]
            else:
                # 没找到关闭标签, 取到下一个标签或末尾
                next_start = container.find("<", m_end)
                if next_start >= 0:
                    el_html = container[start:next_start]
                else:
                    el_html = container[start:]
            results.append((start, tag, el_html))
        elif tag == "img":
            # 自闭合img
            close_pos = container.find(">", m_end)
            if close_pos >= 0:
                el_html = container[start:close_pos + 1]
                results.append((start, tag, el_html))

    return results


def crawl(max_pages=5):
    import sqlite3
    conn = sqlite3.connect(DB_PATH)
    cur = conn.cursor()

    new_count = 0
    skip_count = 0
    error_count = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = BASE_URL + LIST_BASE
        else:
            list_url = BASE_URL + ("%s_%d" % (LIST_BASE, page))

        print("[分页] 第%d页: %s" % (page, list_url), flush=True)

        time.sleep(DELAY)
        html = fetch(list_url)
        if not html:
            print("  [完成] 第%d页无法获取" % page, flush=True)
            break

        items = parse_list(html)
        if not items:
            print("  [完成] 第%d页无数据" % page, flush=True)
            break

        print("  找到 %d 条" % len(items), flush=True)

        for title, detail_url, date in items:
            cur.execute("SELECT id, content FROM gov_raw WHERE page_url=? AND site_name=?",
                       (detail_url, SITE_NAME))
            row = cur.fetchone()

            if row and row[1] and len(row[1]) > 10:
                skip_count += 1
                continue

            time.sleep(DELAY)

            detail_html = fetch(detail_url)
            if not detail_html:
                print("  [等待5s重试] %s" % title[:30], flush=True)
                time.sleep(5)
                detail_html = fetch(detail_url)
                if not detail_html:
                    print("  [跳过] 详情页: %s" % title[:30], flush=True)
                    error_count += 1
                    continue

            content, attachments = parse_detail(detail_html, detail_url)
            attachments_json = json.dumps(attachments, ensure_ascii=False)

            if content:
                summary = content[:200]
                print("  [正文] %s -> %d chars, %d附件" %
                      (title[:30], len(content), len(attachments)), flush=True)
            else:
                summary = title
                print("  [无文本] %s (扫描件/附件页)" % title[:30], flush=True)

            if row:
                cur.execute(
                    "UPDATE gov_raw SET content=?, title=?, attachments=?, summary=? WHERE id=?",
                    (content, title, attachments_json, summary, row[0])
                )
            else:
                cur.execute(
                    """INSERT OR IGNORE INTO gov_raw
                       (site_name, page_url, title, content, publish_date, attachments, summary)
                       VALUES (?, ?, ?, ?, ?, ?, ?)""",
                    (SITE_NAME, detail_url, title, content, date, attachments_json, summary)
                )

            if cur.rowcount > 0 or (row and True):
                new_count += 1
            else:
                skip_count += 1

        conn.commit()

    conn.close()
    print("\n[DONE] %s: 新增=%d, 跳过=%d, 错误=%d" %
          (SITE_NAME, new_count, skip_count, error_count), flush=True)


if __name__ == "__main__":
    max_p = int(sys.argv[1]) if len(sys.argv) > 1 else 5
    crawl(max_p)
