#!/usr/bin/env python3
"""
长治市潞州区人民政府-公示公告 爬虫
CMS: TRS CMS
列表: .../gzdt/gsgg/ (page1), .../gzdt/gsgg/index_N.html (page2+)
详情: .../gzdt/gsgg/YYYYMM/tYYMMDD_NNNNNNN.html
共20页354条, 每页18条
TRS_Editor 正文容器
"""
import os, re, sys, json, time
from bs4 import BeautifulSoup
from urllib import request
from urllib.parse import urljoin
import ssl
import urllib.parse

SITE_NAME = "长治市潞州区人民政府-公示公告"
BASE_URL = "http://www.luzhouqu.gov.cn"
LIST_PATH = "/lzqzw/zfxxgk/zfxxgk/gzdt/gsgg"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
DELAY = 1.5

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

ssl._create_default_https_context = ssl._create_unverified_context


def fetch(url, timeout=30):
    req = request.Request(url, headers=HEADERS)
    try:
        resp = request.urlopen(req, timeout=timeout)
        return resp.read().decode("utf-8", errors="ignore")
    except Exception as e:
        print("  [ERROR] %s: %s" % (url, e), file=sys.stderr)
        return None


def parse_list(html, list_url):
    """从列表页提取文章信息."""
    items = []
    for m in re.finditer(
        r'<li><a href="\./([^"]+)"[^>]*>\s*([^<]+)\s*</a><span[^>]*>\s*(\d{4}-\d{2}-\d{2})\s*</span></li>',
        html
    ):
        href = m.group(1)
        title = m.group(2).strip()
        date = m.group(3)
        if not title or len(title) < 5:
            continue
        # 相对路径 => 绝对URL
        full_url = urljoin(list_url, href)
        items.append((title, full_url, date))
    return items


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def _extract_elements_sorted(container):
    results = []
    for m in re.finditer(r"<(p|table|img)\b[^>]*>", container, re.DOTALL):
        start, m_end, tag = m.start(), m.end(), m.group(1)
        if tag in ("p", "table"):
            close_tag = "</" + tag + ">"
            end_pos = container.find(close_tag, m_end)
            if end_pos >= 0:
                el_html = container[start:end_pos + len(close_tag)]
            else:
                ns = container.find("<", m_end)
                el_html = container[start:ns] if ns >= 0 else container[start:]
            results.append((start, tag, el_html))
        elif tag == "img":
            cp = container.find(">", m_end)
            if cp >= 0:
                results.append((start, tag, container[start:cp + 1]))
    return results


def parse_detail(html, url):
    """提取正文(含表格、图片)和附件. TRS_Editor容器."""
    content_parts = []
    attachments = []

    container = None
    for pat in [
        r'<div[^>]*class="[^"]*TRS_Editor[^"]*"[^>]*>(.*?)</div>',
        r'<div[^>]*class="conTxt"[^>]*>(.*?)</div>',
        r'<div[^>]*class="content"[^>]*>(.*?)</div>',
        r'<div[^>]*class="article"[^>]*>(.*?)</div>',
        r'<div[^>]*class="article-con"[^>]*>(.*?)</div>',
        r'<div[^>]*id="zoom"[^>]*>(.*?)</div>',
    ]:
        m = re.search(pat, html, re.DOTALL)
        if m:
            container = m.group(1)
            break

    # Extract PDF URLs from script tags (PDF-embedded pages)
    pdf_urls = []
    for sm in re.finditer(r'<script[^>]*>.*?pdfurl\s*=\s*[\'"](\./[^\'"]+\.pdf)[\'"].*?</script>', html, re.DOTALL):
        pdf_path = sm.group(1)
        pdf_url = urljoin(url, pdf_path)
        pdf_urls.append(pdf_url)

    if not container:
        for pu in pdf_urls:
            attachments.append({"title": "PDF附件", "url": pu})
        body_m = re.search(r"<body[^>]*>(.*?)</body>", html, re.DOTALL)
        if body_m:
            for p in re.findall(r"<p[^>]*>(.*?)</p>", body_m.group(1), re.DOTALL):
                text = re.sub(r"<[^>]+>", "", p).strip()
                if text:
                    content_parts.append(text)
        return "\n\n".join(content_parts), []

    # table位置
    table_ranges = []
    for tm in re.finditer(r"<table[^>]*>.*?</table>", container, re.DOTALL):
        table_ranges.append((tm.start(), tm.end()))

    def inside_table(pos):
        return any(ts <= pos <= te for ts, te in table_ranges)

    def has_block_children(el_html):
        """Check if a <p> element wraps block-level elements (like <div>, <p>, <table>)."""
        body = re.sub(r'^<p[^>]*>', '', el_html)
        body = re.sub(r'</p>\s*$', '', body)
        return bool(re.search(r'<(div|p|table|ul|ol|h[1-6])\b', body))

    elements = []
    for pos, tag, el_html in _extract_elements_sorted(container):
        if tag == "p":
            if inside_table(pos):
                continue
            if has_block_children(el_html):
                continue
            if "<td" in el_html or "<th" in el_html:
                continue
            text = re.sub(r"<[^>]+>", "", el_html).strip()
            text = re.sub(r"\s+", " ", text)
            if text:
                elements.append((pos, text))
        elif tag == "table":
            if el_html.count("<td") > 0:
                md = html_table_to_html(el_html)
                if md:
                    elements.append((pos, md))
        elif tag == "img":
            src_m = re.search(r'src="([^"]+)"', el_html)
            alt_m = re.search(r'alt="([^"]*)"', el_html)
            if src_m:
                src = src_m.group(1)
                full_src = urljoin(url, src)
                alt = alt_m.group(1) if alt_m else ""
                elements.append((pos, "![%s](%s)" % (alt, full_src)))

    elements.sort(key=lambda x: x[0])
    content = "\n\n".join(text for _, text in elements)

    # 附件
    for a_href, a_text in re.findall(
        r'<a[^>]*href="([^"]+\.(?:doc|docx|pdf|xls|xlsx|ppt|pptx|zip|rar|7z|txt|wps|et))"[^>]*>([^<]+)</a>',
        html, re.DOTALL
    ):
        attachments.append({"title": a_text.strip(), "url": urljoin(url, a_href)})

    # Also add PDF URLs extracted from script tags
    seen_urls = {a['url'] for a in attachments}
    added_pdfs = []
    for sm in re.finditer(r'<script[^>]*>.*?pdfurl\s*=\s*[\'"](\./[^\'"]+\.pdf)[\'"].*?</script>', html, re.DOTALL):
        pdf_path = sm.group(1)
        pdf_url = urljoin(url, pdf_path)
        if pdf_url not in seen_urls:
            seen_urls.add(pdf_url)
            pdf_name = pdf_url.split('/')[-1]
            added_pdfs.append(pdf_url)
            attachments.append({"title": "PDF附件: " + pdf_name, "url": pdf_url})

    # If content is empty but we have PDF attachments, generate note
    if (not content or not content.strip()) and added_pdfs:
        content = "【该文档为PDF文件，正文内容见附件PDF链接】\n"
        for pu in added_pdfs:
            content += "PDF下载: " + pu + "\n"

    return content, attachments


def crawl(max_pages=5):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()
    new_count = skip_count = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = BASE_URL + LIST_PATH + "/"
        else:
            # 0-based: page 2 = index_1, page 3 = index_2
            list_url = BASE_URL + LIST_PATH + "/index_%d.html" % (page - 1)

        print("[分页] 第%d页: %s" % (page, list_url), flush=True)
        time.sleep(DELAY)
        html = fetch(list_url)
        if not html:
            print("  [完成] 无法获取", flush=True)
            break

        items = parse_list(html, list_url)
        if not items:
            print("  [完成] 无数据", flush=True)
            break

        print("  找到 %d 条" % len(items), flush=True)

        for title, detail_url, date in items:
            cur.execute("SELECT id FROM gov_raw WHERE page_url=? AND site_name=?", (detail_url, SITE_NAME))
            if cur.fetchone():
                skip_count += 1
                continue

            time.sleep(DELAY)
            detail_html = fetch(detail_url)
            if not detail_html:
                skip_count += 1
                continue

            content, attachments = parse_detail(detail_html, detail_url)
            aj = json.dumps(attachments, ensure_ascii=False)
            summary = content[:200] if content else title

            cur.execute(
                "INSERT OR IGNORE INTO gov_raw (site_name,page_url,title,content,publish_date,attachments,summary) VALUES(?,?,?,?,?,?,?)",
                (SITE_NAME, detail_url, title, content, date, aj, summary)
            )
            if cur.rowcount > 0:
                new_count += 1
                if content:
                    print("  [正文] %s -> %d chars, %d附件" %
                          (title[:35], len(content), len(attachments)), flush=True)
                else:
                    print("  [无文本] %s" % title[:35], flush=True)

        conn.commit()

    conn.close()
    print("\n[DONE] %s: 新增=%d, 跳过=%d" % (SITE_NAME, new_count, skip_count), flush=True)


if __name__ == "__main__":
    crawl(int(sys.argv[1]) if len(sys.argv) > 1 else 5)
