#!/usr/bin/env python3
"""建平县人民政府-环保信息公开 爬虫
CMS: 朝阳CMS (cms.chaoyang.gov.cn)
列表: /jpxzf/zwgk/hbxxgk/hbxxgk/glist.html (page1), glist{N}.html (page2+ via redirect)
分页: 29页 x 15条
详情: /html/JPXZF/YYYYMM/0{objectId}.html, 容器 div.center-info
"""
import os, re, sys, json, time
from bs4 import BeautifulSoup
from urllib import request
from urllib.parse import urljoin
import ssl
import urllib.parse

SITE_NAME = "建平县人民政府-环保信息公开"
BASE_URL = "https://www.lnjp.gov.cn"
LIST_PATH = "/jpxzf/zwgk/hbxxgk/hbxxgk/glist"
DB_PATH = os.getenv("SEARCH_DB", "/root/search.db")
DELAY = 2

HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0.0.0 Safari/537.36",
    "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
    "Accept-Language": "zh-CN,zh;q=0.9",
}

ssl._create_default_https_context = ssl._create_unverified_context


def fetch(url, timeout=30):
    req = request.Request(url, headers=HEADERS)
    try:
        resp = request.urlopen(req, timeout=timeout)
        return resp.read().decode("utf-8", errors="ignore")
    except Exception as e:
        print("  [ERROR] %s: %s" % (url, e), file=sys.stderr)
        return None


def parse_list(html):
    """提取列表项: li > a + span(date)"""
    items = []
    for m in re.finditer(
        r'<li>\s*<a[^>]*href="([^"]+)"[^>]*>([^<]+)</a>\s*<span>(\d{4}-\d{2}-\d{2})</span>',
        html, re.DOTALL
    ):
        href = m.group(1).strip()
        title = m.group(2).strip()
        date = m.group(3)
        if not title or len(title) < 5:
            continue
        full_url = href if href.startswith("http") else (BASE_URL + href)
        items.append((title, full_url, date))
    return items


def html_table_to_html(table, base_url=""):
    """保留 HTML 表格结构，仅将相对链接/图片转绝对 URL"""
    from bs4 import BeautifulSoup
    tbl = BeautifulSoup(str(table), 'html.parser')
    for a in tbl.find_all('a'):
        href = a.get('href', '')
        if href and not href.startswith(('http', 'javascript', '#')):
            a['href'] = urllib.parse.urljoin(base_url, href) if base_url else href
    for img in tbl.find_all('img'):
        src = img.get('src', '')
        if src and not src.startswith(('http', '//', 'data:')):
            img['src'] = urllib.parse.urljoin(base_url, src) if base_url else src
    return str(tbl)


def parse_detail(html, url):
    content_parts = []
    attachments = []

    # 正文容器: div.center-info
    container = None
    for pat in [
        r'<div\s+class="center-info"[^>]*>(.*?)</div>\s*</div>',
        r'<div\s+class="center-info"[^>]*>(.*?)</div>',
    ]:
        m = re.search(pat, html, re.DOTALL)
        if m:
            container = m.group(1)
            break

    if container:
        # 提取所有p段落
        for p_html in re.findall(r'<p[^>]*>(.*?)</p>', container, re.DOTALL):
            text = re.sub(r"<[^>]+>", "", p_html).strip()
            text = text.replace("&nbsp;", " ")
            text = re.sub(r"\s+", " ", text)
            # 检查附件链接
            for a_href, a_text in re.findall(
                r'<a[^>]*href="([^"]+\.(?:docx?|pdf|xlsx?|rar|zip|wps|et))"[^>]*>([^<]+)</a>',
                p_html, re.I
            ):
                full = urljoin(url, a_href)
                if a_text.strip():
                    attachments.append({"title": a_text.strip(), "url": full})
            if text and text.strip():
                content_parts.append(text)

        # 提取表格
        for t_html in re.findall(r'<table[^>]*>.*?</table>', container, re.DOTALL):
            table_md = html_table_to_html(t_html)
            if table_md:
                content_parts.append(table_md)
    else:
        body_m = re.search(r"<body[^>]*>(.*?)</body>", html, re.DOTALL)
        if body_m:
            for p_html in re.findall(r"<p[^>]*>(.*?)</p>", body_m.group(1), re.DOTALL):
                text = re.sub(r"<[^>]+>", "", p_html).strip()
                text = text.replace("&nbsp;", " ")
                text = re.sub(r"\s+", " ", text)
                if text:
                    content_parts.append(text)

    # 全文附件补充
    for a_href, a_text in re.findall(
        r'<a[^>]*href="([^"]+\.(?:docx?|pdf|xlsx?|rar|zip))"[^>]*>([^<]+)</a>',
        html, re.I
    ):
        full = urljoin(url, a_href)
        if not any(a["url"] == full for a in attachments):
            attachments.append({"title": a_text.strip(), "url": full})

    content = "\n\n".join(content_parts)
    return content, attachments


def get_title_from_html(html):
    m = re.search(r'<div\s+class="center-title">([^<]+)</div>', html)
    if m:
        return m.group(1).strip()
    return None


def crawl(max_pages=5):
    import sqlite3
    conn = sqlite3.connect(DB_PATH, timeout=60)
    cur = conn.cursor()

    new_count = 0
    skip_count = 0
    error_count = 0

    for page in range(1, max_pages + 1):
        if page == 1:
            list_url = BASE_URL + LIST_PATH + ".html"
        else:
            list_url = BASE_URL + LIST_PATH + str(page - 1) + ".html"

        print("[分页] 第%d页: %s" % (page, list_url), flush=True)

        time.sleep(DELAY)
        html = fetch(list_url)
        if not html:
            print("  [完成] 第%d页无法获取" % page, flush=True)
            break

        items = parse_list(html)
        if not items:
            print("  [完成] 第%d页无数据" % page, flush=True)
            break

        print("  找到 %d 条" % len(items), flush=True)

        for title, detail_url, date in items:
            cur.execute("SELECT id, content FROM gov_raw WHERE page_url=? AND site_name=?",
                       (detail_url, SITE_NAME))
            row = cur.fetchone()

            if row and row[1] and len(row[1]) > 10:
                skip_count += 1
                continue

            time.sleep(DELAY)

            detail_html = fetch(detail_url)
            if not detail_html:
                print("  [等待5s重试] %s" % title[:30], flush=True)
                time.sleep(5)
                detail_html = fetch(detail_url)
                if not detail_html:
                    print("  [跳过] 详情页: %s" % title[:30], flush=True)
                    error_count += 1
                    continue

            content, attachments = parse_detail(detail_html, detail_url)
            attachments_json = json.dumps(attachments, ensure_ascii=False)

            detail_title = get_title_from_html(detail_html)
            if detail_title:
                title = detail_title

            if content:
                summary = content[:200]
                print("  [正文] %s -> %d chars, %d附件" %
                      (title[:30], len(content), len(attachments)), flush=True)
            else:
                summary = title
                print("  [无文本] %s (扫描件)" % title[:30], flush=True)

            if row:
                cur.execute(
                    "UPDATE gov_raw SET content=?, title=?, attachments=?, summary=? WHERE id=?",
                    (content, title, attachments_json, summary, row[0])
                )
            else:
                cur.execute(
                    """INSERT OR IGNORE INTO gov_raw
                       (site_name, page_url, title, content, publish_date, attachments, summary)
                       VALUES (?, ?, ?, ?, ?, ?, ?)""",
                    (SITE_NAME, detail_url, title, content, date, attachments_json, summary)
                )

            if cur.rowcount > 0 or (row and True):
                new_count += 1
            else:
                skip_count += 1

        conn.commit()

    conn.close()
    print("\n[DONE] %s: 新增=%d, 跳过=%d, 错误=%d" %
          (SITE_NAME, new_count, skip_count, error_count), flush=True)


if __name__ == "__main__":
    max_p = int(sys.argv[1]) if len(sys.argv) > 1 else 5
    crawl(max_p)
