#!/usr/bin/env python3
"""
crawl_sdlanyi.py - 山东蓝一检测技术有限公司-公示信息
列表: http://www.sdlanyi.com/html/gongshixinxi/ (index_N.html 分页, 47页共702条)
  条目: <dd><a title="标题" href="/html/gongshixinxi/NNNN.html">标题</a><span class="hui">[日期]</span></dd>
详情: /html/gongshixinxi/NNNN.html, 正文容器 class=wz_nr, h1标题, 附件: 附件下载：<a href="/Upfile/...">PDF</a>
编码: GBK
"""

import sys, os, requests, re, json
sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))
from crawler_lib import push_to_searchdb

BASE = "http://www.sdlanyi.com"
LIST_BASE = BASE + "/html/gongshixinxi/"
SITE_NAME = "山东蓝一检测-公示信息"
HEADERS = {
    "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 Chrome/120.0",
}


def fetch_list(page=1):
    if page == 1:
        url = LIST_BASE
    else:
        url = f"{LIST_BASE}index_{page}.html"
    try:
        r = requests.get(url, headers=HEADERS, timeout=25)
        r.encoding = "gbk"
        html = r.text
    except Exception as e:
        print(f"  [LIST ERR] {e}")
        return [], 0
    items = []
    for m in re.finditer(r'<a\s+title="([^"]{4,300})"\s+href="(/html/gongshixinxi/\d+\.html)"[^>]*>(?:[^<]*)</a></span><span class="hui">\[(\d{4}-\d{2}-\d{2})', html):
        items.append({"url": BASE + m.group(2), "title": m.group(1).strip(), "date": m.group(3)})
    # 兜底: 匹配不带 title 属性的
    if not items:
        for m in re.finditer(r'<a\s+title="([^"]{4,300})"\s+href="(/html/gongshixinxi/\d+\.html)"', html):
            items.append({"url": BASE + m.group(2), "title": m.group(1).strip(), "date": ""})
    # 总页数
    total_pages = 0
    m = re.search(r"页次:(\d+)/(\d+)", html)
    if m:
        total_pages = int(m.group(2))
    return items, total_pages


def fetch_detail(url):
    try:
        r = requests.get(url, headers=HEADERS, timeout=25)
        r.encoding = "gbk"
        html = r.text
    except Exception as e:
        print(f"  [DETAIL ERR] {e}")
        return "", "", ""
    # 标题 h1 (wz_nr 前)
    title = ""
    m = re.search(r"<h1[^>]*>([^<]{4,300})</h1>", html)
    if m:
        title = m.group(1).strip()
    # 正文 wz_nr
    content = ""
    idx = html.find('class="wz_nr"')
    if idx > 0:
        # 找到容器内第一个 > 之后
        start = html.find(">", idx) + 1
        # 匹配到对应 </div> (简单: 找下一个 "附件下载" 或容器尾)
        end = html.find("</div>", start)
        if end > 0:
            content = html[start:end].strip()
    # 附件绝对化 + 清洗
    content = re.sub(r'href="/', 'href="' + BASE + '/', content)
    content = re.sub(r'src="/', 'src="' + BASE + '/', content)
    # 附件链接独立成段 <p><a> (用户偏好)
    content = re.sub(r'(附件下载：)(<a[^>]*>.*?</a>)', r'<p>\1\2</p>', content, flags=re.S)
    # 日期: 从正文提取 公示时间
    pub_date = ""
    m = re.search(r"公示时间[：:]\s*(\d{4})[年\-/](\d{1,2})[月\-/](\d{1,2})", content)
    if m:
        pub_date = f"{m.group(1)}-{int(m.group(2)):02d}-{int(m.group(3)):02d}"
    return title, pub_date, content


def main():
    import argparse
    parser = argparse.ArgumentParser()
    parser.add_argument("--pages", type=int, default=1)
    parser.add_argument("--all", action="store_true")
    args = parser.parse_args()

    items_all = []
    seen = set()
    first_items, total_pages = fetch_list(1)
    print(f"  总页数: {total_pages}, 首页 {len(first_items)} 条")
    if args.all:
        max_pages = total_pages if total_pages else 47
    else:
        max_pages = args.pages
    for pg in range(1, max_pages + 1):
        if pg > 1:
            items, _ = fetch_list(pg)
        else:
            items = first_items
        print(f"  第{pg}页: {len(items)} 条")
        for it in items:
            if it["url"] not in seen:
                seen.add(it["url"])
                items_all.append(it)

    records = []
    for idx, it in enumerate(items_all):
        print(f"  [{idx+1}/{len(items_all)}] {it['title'][:40]}...")
        t, d, content = fetch_detail(it["url"])
        if not content.strip():
            print("    [SKIP] 正文空")
            continue
        records.append({
            "title": t or it["title"],
            "url": it["url"],
            "pub_date": d or it["date"],
            "site_name": SITE_NAME,
            "content": content,
            "summary": "",
        })
    print(f"  共 {len(records)} 条有效")
    if records:
        push_to_searchdb(records, "sdlanyi")
    return len(records)


if __name__ == "__main__":
    cnt = main()
    print(f"Done: {cnt} records")
