#!/usr/bin/env python3

"""

统一搜索服务（端口8000）

集成：gov FTS5搜索 + 中项网/中能联合/中策大数据

"""



import http.server, sqlite3, urllib.parse, re, os, html, json, datetime, io, csv, signal, sys, markdown, uuid, string




def contains_chinese(text):
    return bool(re.search(r"[\u4e00-\u9fff\u3400-\u4dbf]", text))


DB_PATH = "/root/search.db"

CONTACT_DB_PATH = "/root/ccpc.db"

# 2026-09-10: 统一联系人库(contact_lib.py 从 ccpc.db+ZC.db 聚合) — /contact 页专用;
# CONTACT_DB_PATH(ccpc.db) 仍服务 /ccpc 项目浏览与项目详情联系人
CONTACT_ROUTE_DB = "/root/contact.db"

# 角色展示归一/过滤(2026-09-10 用户裁定): '业主/业主单位/owner' 是硬模板默认值不显示;
# 英文别名归一到中文; 过滤后为空则不渲染"角色"字段
CONTACT_TEMPLATE_ROLES = {'业主', '业主单位', '业主方', 'owner', 'ref'}
CONTACT_ROLE_ALIAS = {'owner': '业主', 'design': '设计院', 'contractor': '施工单位'}


def contact_role_display(raw):
    parts = [p.strip() for p in (raw or '').split(' / ') if p.strip()]
    out = []
    for p in parts:
        v = CONTACT_ROLE_ALIAS.get(p.lower(), p)
        if v in CONTACT_TEMPLATE_ROLES or v.lower() in CONTACT_TEMPLATE_ROLES:
            continue
        if v not in out:
            out.append(v)
    return ' / '.join(out)

ZNLH_DB_PATH = "/home/ccbuild/znlh.db"

ZC_DB_PATH = "/root/ZC.db"

EIA_DB_PATH = "/root/eia.db"

CSV_PATH = "/root/data.csv"

PORT = 8000



def extract_investment(text):

    if not text: return '-'

    m = re.search(r'[总]?投资[额金]?[：:为]?\s*[约]?\s*[\d,.]+\s*[万亿千百十万千百元美]?[元美]?', text)

    if m: return m.group(0).strip()

    m = re.search(r'投资.{0,6}?[\d,.]+[万亿]?', text)

    if m: return m.group(0).strip()

    return '-'



PHASES = [

    {'id':'eia','label':'环评阶段','keywords':'环评 OR 环境影响评价 OR 环境影响报告 OR 环境影响登记 OR 第一次公示 OR 第二次公示 OR 征求意见 OR 公众参与'},

    {'id':'approval','label':'审批报批','keywords':'审批 OR 报批 OR 批复 OR 核准 OR 备案 OR 行政许可 OR 审查意见 OR 同意建设'},

    {'id':'bidding','label':'招标采购','keywords':'招标 OR 中标 OR 采购 OR 比选 OR 邀标 OR 竞标 OR 竞争性谈判 OR 询价 OR 招标公告 OR 中标候选人'},

    {'id':'acceptance','label':'竣工验收','keywords':'验收 OR 竣工 OR 试运行 OR 投产 OR 调试 OR 环境保护验收 OR 竣工验收'},

    {'id':'planning','label':'规划设计','keywords':'规划 OR 设计 OR 可行性研究 OR 可研 OR 选址 OR 用地预审 OR 方案设计 OR 初步设计'},

    {'id':'security','label':'安全评价','keywords':'安全评价 OR 安全设施 OR 安全验收 OR 安全预评价 OR 安全专篇 OR 安全生产 OR 风险评估'},

]

PHASE_MAP = {p['id']: p for p in PHASES}



# ─── 行业分类（一级标签，仅用标题匹配）───

INDUSTRY_RULES = [
    ('pharma', '制药与生物技术', ['APIs','API','GMP','cGMP','疫苗','生物制品','干细胞','抗体','基因','诊断试剂','医疗器械', '药品', '制药', '医药', '药业','医药','胶原蛋白','单克隆','血浆','胰岛素','造影剂','缝合线','导管','内窥镜','起搏器','透析','神经外科','片剂','丸剂','胶囊','软膏','注射剂','注射器','口服液','辅料','CDMO','多肽', '原药', '母药', '原料药', '兽药', '制剂', '中成药', '生物医药', '制药项目', '医药中间体', '原料药及', '药厂', '制药厂', '医药产业园', '药用']),
    ('production', '生产/开采', ['油田','气田','采油','采气','钻井','页岩气','海上平台','酸气处理','天然气处理站','天然气集气站','天然气生产平台','煤层气生产平台']),
    ('transmission', '管道输送', ['管道','输气','输油','门站','计量站','压气站','阀室','装卸站及专用铁路']),
    ('terminals', '终端/接收站', ['接收站','LNG终端','LNG','天然气终端','天然气储存终端','油库','罐区','原油储罐','成品油罐','储运','成品油仓储码头','电煤港','煤炭港', '油品码头', '原油码头', '成品油码头', 'LNG码头', '液化天然气码头', '天然气码头', '油气码头', '液体化工码头', '加油站', '加气站', '加油站扩建', '成品油销售', '燃气公司', '燃气供应', '液化气站', '充电站', '换电站']),
    ('port', '港口码头', ['港口','码头','泊位','港区','港务','集装箱','散货','件杂货','客滚','轮渡','航运','疏浚','航道','船闸','引航','渡口','装卸']),
    ('hpi', '石油炼制', ['炼化一体','炼化','炼油','炼厂','石脑油','催化裂化','加氢裂化','FCC']),
    ('altfuel', '替代燃料', ['生物柴油','生物乙醇','可持续航空燃料','沼气','燃料电池','乙醇燃料','甲醇燃料','SAF', '生物质颗粒', '生物质燃料', '生物质成型', '生物质能', '生物质气化']),
    ('power', '电力', ['发电','电站','光伏','风电','太阳能','储能','BESS','CAES','CSP','变电站','变电所','输变电','输电','电网','核电','水电站','抽水蓄能','压缩空气储能','煤电','供热', '生物质发电', '生物质锅炉', '生物质热电']),
    ('metals', '金属与矿物', ['金属与矿物','有色金属','稀土','稀有金属','不锈钢','钢铁','冶金','矿山','采矿','选矿','尾矿','冶炼','钢厂','轧钢','炼铁','炼钢','铁合金','钢铁厂','金矿','铁矿','铜矿','铝土矿','磷矿','石墨矿','石墨化电极','石墨','钢结构','石料','多金属矿','锌矿','铅矿','镍矿','镁合金','铝合金','钛合金','铜合金','锌合金','轻合金','铝板','铝业','铜业','锌业','金属','矿物','铝','铜','锌','铅','镍','锡','钨','钼','合金','矿','水泥','预拌混凝土','建材','石材','瓷砖','石膏板','防水材料','保温材料','耐火材料','马口铁','易拉罐', '混凝土', '砂浆', '商砼', '沥青混凝土', '水泥制品', '搅拌站', '预拌砂浆']),
    ('pulp', '制浆/造纸/木材', ['纸浆','造纸','纸板','木材','木业','人造板','瓦楞','刨花板','胶合板','纸厂','印刷品', '印刷包装','纸包装']),
    ('food', '食品与饮料', ['食品','饮料','肉类','屠宰','乳','牛奶','果汁','啤酒','调味','面包','豆制品','食用油','饲料','糖','淀粉','酱油','酵母','茶','咖啡','水产','禽','畜','碾米','酒厂','烈酒','植物产品','瓶装水','矿泉水','烘焙','花生加工','鸡蛋加工','鱼片','香料', '养殖', '生猪', '蛋鸡', '肉鸭', '肉鸡', '水产养殖', '酿酒', '白酒', '啤酒厂', '酱', '预制菜', '果干', '果汁厂', '肉类加工', '屠宰场', '饲料厂']),
    ('cpi', '化工加工', ['废矿物油','化工','化学品','树脂','聚合物','聚乙烯','聚丙烯','PVC','PET','PTA','ABS','EVA','HDPE','LDPE','PE管','PPR','塑料','橡胶','涂料','染料','肥料','农药','溶剂','催化剂','酸','酯','醇','硫酸','盐酸','氢氧化','氧化物','丙烯','乙烯','苯','酮','醚','胺','烯','硅','氟','氯','溴','碘','粘合剂','添加剂','阻燃剂','表面活性剂','精细化学品','水处理剂','复合肥','水溶肥','颜料','洗涤剂','增塑剂','活性炭','高纯度化学品','湿化学品','过氧化氢','甲醛','甲醇','氨','氧气','空气分离','烧碱','化学试剂','化学中间体','炭黑','气凝胶','脱硫剂','纺织助剂','纤维素纤维','紫外线吸收剂','电解质','杀菌剂','杀虫剂','除草剂','制冷剂','减水剂','乙炔','双酚','莱赛尔纤维','脂肪','炔','腈','酐','醛','酚','氢能','绿氢','氢气','电解水制氢','电解制氢','水电解制氢','制氢','氟化石墨','全氟聚醚','氟化改性','高纯氟气','软包装','复合膜', '植物保护', '种子处理剂', '农药制剂', '农药中间体', '植物蛋白', '生物农药', '农药原药']),
    ('logistics', '物流', ['物流','物流中心','物流园','仓储物流','保税物流','冷链物流','快递分拨']),
    ('manufacturing', '工业制造', ['建筑用砂','制造','设备','机械','零部件','电子','汽车','半导体','电池','家电','仪器','配件','泵','压缩机','阀门','轴承','传感器','显示屏','印刷电路板', '印刷线路板','连接器','变压器','无人机','模具','晶圆','数控机床','造船','起重机','空调','触摸屏','芯片','机器人','雷达','集成电路','光刻胶','电机','锅炉','过滤器','船厂','机床','紧固件','风机','飞机','数据中心','家具','包装', '电缆', '电线电缆', '特种电缆', '面料', '染整', '织造', '针织', '梭织', '制鞋', '鞋', '标签', '不干胶', '五金', '五金制品', '制品', '电子元件', '光学']),
]

INDUSTRY_MAP = {ind_id: {'id': ind_id, 'label': label, 'kw': kws} for ind_id, label, kws in INDUSTRY_RULES}

# 预计算关键词对（按长度降序），避免 classify_industry 每次调用重复构建+排序
CLASSIFY_PAIRS = sorted(
    ((len(kw), ind_id, kw.lower()) for ind_id, label, kws in INDUSTRY_RULES for kw in kws),
    key=lambda x: -x[0],
)

# 行政公告排除（2026-08-11 用户决策: 行政公告保持 other）
ADMIN_EXCLUDE = [
    '医保', '药店', '医药机构', '定点零售', '零售药店', '医保服务协议',
    '医保定点', '集采药品销售', '药品经营', '药品价格', '门诊慢性病',
    '医疗保障', '医保目录', '药师', '执业药师',
]

def classify_industry(title):
    """仅用标题做一级行业分类，返回行业 id。最长关键词优先（用户体系）。
    小写匹配：英文关键词大小写不敏感（BESS/bess/Bess 均命中）。"""
    if not title: return 'other'
    t = title.lower()
    # 行政公告排除: 医保/药店/定点零售等 → other (用户决策 2026-08-11)
    for kw in ADMIN_EXCLUDE:
        if kw.lower() in t:
            return 'other'
    # 地名等误伤排除（命中即不是 metals）
    if '铜梁' in t or '铜仁' in t:
        t = t.replace('铜梁', '·').replace('铜仁', '·')
    # 用户2026-08-23: 包含'制品' → 制造业, 除非命中已有的行业特定制品词(生物制品/水泥制品/豆制品/五金制品等)
    if '制品' in t:
        existing_goods = [kw for ind_id, label, kws in INDUSTRY_RULES for kw in kws if kw.endswith('制品') and len(kw) >= 3]
        if not any(kw in t for kw in existing_goods):
            return 'manufacturing'
    for _len, ind_id, kw in CLASSIFY_PAIRS:
        if kw in t:
            return ind_id
    return 'other'

def ensure_industry_column():
    """启动时确保 gov_raw 有 industry 列"""
    try:
        conn = sqlite3.connect(DB_PATH, timeout=10)
        conn.execute("ALTER TABLE gov_raw ADD COLUMN industry TEXT DEFAULT 'other'")
        conn.commit()
        conn.close()
    except Exception:
        pass  # 列已存在则忽略



CATEGORIES = [

    {'id': '资源网站', 'label': '资源网站'},

]



def esc(s):

    if s is None: return ''

    return str(s).replace('&','&amp;').replace('<','&lt;').replace('>','&gt;').replace('"','&quot;')

def render_md(text):
    """Render Markdown, ensuring blank lines before tables for proper table detection"""
    if not text: return ''
    # Ensure blank line before table: a line containing ' | ' followed by line starting with '---'
    text = re.sub(r'(?<!\n\n)(\n)([^\n]* \| [^\n]*\n)(--- \|)', r'\n\n\2\3', text)
    return markdown.markdown(text, extensions=['extra'])


CSS_STYLE = """*{margin:0;padding:0;box-sizing:border-box}

body{font-family:-apple-system,BlinkMacSystemFont,"Segoe UI","Microsoft YaHei",sans-serif;background:#fff;color:#202124;min-height:100vh}

a{color:#1a0dab;text-decoration:none}

a:visited{color:#609}

a:hover{text-decoration:underline}

.container{max-width:1000px;margin:0 auto;padding:14px 20px 56px}

.date-chip{display:inline-block;padding:3px 12px;border:1px solid #dadce0;border-radius:14px;font-size:13px;color:#1a73e8;text-decoration:none;cursor:pointer;background:#fff}
.date-chip:hover{background:#e8f0fe;border-color:#1a73e8;text-decoration:none}
.date-chip.act{background:#1a73e8;color:#fff;border-color:#1a73e8;font-weight:500}

.home-wrap{display:flex;flex-direction:column;align-items:center;justify-content:center;min-height:75vh;text-align:center}

.home-logo{font-size:48px;font-weight:700;color:#1a73e8;letter-spacing:-1px;margin-bottom:4px}

.home-sub{font-size:14px;color:#5f6368;margin-bottom:28px}

.home-search{width:100%;max-width:584px;margin:0 auto 24px}

.home-search .search-box input{border-radius:28px;padding:14px 20px;font-size:16px;box-shadow:0 1px 6px rgba(32,33,36,.08)}

.home-search .search-box input:focus{box-shadow:0 1px 8px rgba(32,33,36,.16)}

.home-btns{display:flex;gap:10px;flex-wrap:wrap;justify-content:center;margin-bottom:20px}

.home-btns a{display:inline-flex;align-items:center;gap:6px;padding:8px 20px;background:#f8f9fa;border:1px solid #dadce0;border-radius:24px;font-size:14px;color:#3c4043;transition:all .15s}

.home-btns a:hover{background:#e8f0fe;border-color:#1a73e8;color:#1a73e8;text-decoration:none}

.home-count{font-size:13px;color:#70757a;margin-top:4px}

.nav{display:flex;align-items:center;gap:16px;padding:12px 0;border-bottom:1px solid #ebebeb;margin-bottom:16px}

.nav-logo{font-size:18px;font-weight:700;color:#1a73e8;white-space:nowrap;flex-shrink:0}

.nav-links{display:flex;gap:4px;flex-wrap:wrap}

.nav-links a{font-size:13px;color:#5f6368;padding:4px 10px;border-radius:16px;transition:background .15s}

.nav-links a:hover{background:#f0f0f0;text-decoration:none;color:#1a73e8}

.nav-links a.act{background:#e8f0fe;color:#1a73e8;font-weight:500}

/* .nav-right 导航簇样式由 nav_right_html() 自带（自包含 <style>），此处不再重复定义 */

.topbar{display:flex;align-items:center;padding:12px 0 0}

.search-section{margin-bottom:12px}

.search-box{display:flex;gap:0}

.search-box input{flex:1;padding:12px 20px;border:1px solid #dfe1e5;border-radius:24px 0 0 24px;font-size:16px;outline:none;box-shadow:0 1px 6px rgba(32,33,36,.08)}

.search-box input:focus,.search-box input:hover{border-color:#1a73e8;box-shadow:0 1px 8px rgba(32,33,36,.16)}

.search-box button{padding:12px 24px;background:#1a73e8;color:#fff;border:none;border-radius:0 24px 24px 0;font-size:15px;cursor:pointer}

.search-box button:hover{background:#1557b0}

.toolbar{margin-bottom:10px}

.controls{display:flex;justify-content:space-between;align-items:center}

.total-count{font-size:14px;color:#70757a}

.db-tabs{display:flex;gap:6px;margin-bottom:12px;flex-wrap:wrap}

.db-tabs a{font-size:13px;color:#5f6368;padding:5px 14px;border:1px solid #dadce0;border-radius:16px;transition:all .15s}

.db-tabs a:hover{background:#e8f0fe;text-decoration:none;border-color:#1a73e8;color:#1a73e8}

.db-tabs a.act{background:#1a73e8;color:#fff;border-color:#1a73e8}

.results{max-width:100%;margin:0 auto}.result-item{display:flex;gap:10px;padding:12px 0;border-bottom:1px solid #f0f0f0;transition:background .1s}

.result-item:hover{background:#f8f9fa;margin:0 -10px;padding:12px 10px;border-radius:4px}

.result-item .num{font-size:13px;color:#70757a;min-width:26px;text-align:right;flex-shrink:0;padding-top:2px}

.result-item h3{font-size:16px;font-weight:400;line-height:1.3;margin-bottom:2px}

.result-item h3 a{color:#1a0dab}

.meta{font-size:12px;color:#70757a;display:flex;gap:8px;flex-wrap:wrap;align-items:center}

.bdg{display:inline-block;padding:1px 6px;border-radius:3px;font-size:11px;font-weight:500}

.bdg-crawler{background:#e8f5e9;color:#2e7d32}

.bdg-zc{background:#fef3e2;color:#e65100}

.bdg-other{background:#f3e8fd;color:#7b1fa2}

.orig-id{display:inline-block;padding:1px 6px;border-radius:3px;font-size:11px;font-weight:600;background:#fff3e0;color:#e65100;cursor:help}
.orig-id.comp{background:#f3e8fd;color:#7b1fa2;font-weight:500}

.summary{font-size:14px;color:#4d5156;line-height:1.58;margin-top:2px}

.no-results{text-align:center;padding:60px 0;color:#70757a;font-size:14px}

.time-select{padding:5px 12px;font-size:13px;color:#3c4043;border:1px solid #dadce0;border-radius:16px;cursor:pointer;outline:none;background:#fff}

.time-select:hover{border-color:#bdc1c6}

.page-btn{display:inline-flex;align-items:center;justify-content:center;min-width:36px;height:36px;padding:0 12px;border-radius:4px;font-size:13px;color:#1a73e8;text-decoration:none}

.page-btn:hover{background:#f0f0f0;text-decoration:none}

.page-btn.act{background:#1a73e8;color:#fff}

.page-elp{color:#70757a;padding:0 4px;font-size:13px}

.page-jump{display:inline-flex;align-items:center;gap:4px;margin-left:6px;font-size:13px;color:#5f6368}

.page-jump input{width:48px;padding:6px 8px;border:1px solid #dadce0;border-radius:4px;font-size:13px;text-align:center;outline:none}

.page-jump input:focus{border-color:#1a73e8}

.detail-wrap{max-width:900px;margin:0 auto;padding:20px}

.detail-card{background:#fff;border-radius:12px;padding:24px;margin-bottom:16px;box-shadow:0 1px 3px rgba(0,0,0,.08)}

.detail-card h2{font-size:20px;margin-bottom:12px;color:#1a1a2e}

.detail-content{font-size:14px;line-height:1.8;color:#444;overflow-wrap:break-word;overflow-x:auto}

.detail-content p{margin:10px 0}

.detail-content table{border-collapse:collapse;width:100%;margin:12px 0;font-size:14px}

.detail-content td,.detail-content th{border:1px solid #ddd;padding:8px 12px}

.detail-content a{color:#1a73e8}

.detail-content img{max-width:100%;height:auto;border-radius:8px}

/* ── 2026-09-11 移动端断点（首页 + 搜索结果页是本站唯一没有断点的页面）──
   修复实测缺陷：375px 下 .nav-right（247px 宽）造成横向溢出 6px、
   .nav-links a 被挤到高 80~98px（导航文字逐字换行）。 */
@media(max-width:760px){
.container{padding:10px 12px 40px}
.nav{flex-wrap:wrap;gap:8px;padding:8px 0;margin-bottom:12px}
.nav-logo{font-size:17px}
.nav-links{gap:2px;width:100%}
.nav-links a{font-size:12.5px;padding:5px 9px;white-space:nowrap;flex:0 0 auto}
.topbar{padding:8px 0 0}
h1{font-size:20px}
.home-logo{font-size:34px}
.home-sub{margin-bottom:20px}
.home-wrap{min-height:60vh}
.search-box input{font-size:16px;padding:12px 15px}
.search-box button{padding:12px 16px;font-size:14px}
.controls{flex-direction:column;align-items:flex-start;gap:8px}
.total-count{font-size:13px;line-height:1.6}
.date-bar{flex-wrap:wrap;gap:4px}
.date-chip{font-size:12px;padding:4px 9px}
.result-item{padding:10px 0}
.result-item .num{min-width:20px}
.result-item h3{font-size:15px}
.meta{font-size:11.5px;gap:6px}
.pagination{padding:12px 10px;gap:4px;margin:20px 0 44px}
.page-btn{min-width:34px;height:34px;padding:0 9px;font-size:12.5px}
.page-jump{margin-left:0}
.pager-bar{padding:10px 12px}
}

"""



def token_grams(t):
    """提取 token 的中文连续段的 2-gram 列表（标点/数字/字母作为分隔符跳过）。
    返回空列表 = 该 token 无有效中文 gram（如纯英文/单字中文），无法用 bigram 索引。"""
    segs = re.findall(r'[\u4e00-\u9fff]+', t)
    grams = []
    for seg in segs:
        if len(seg) == 1:
            continue
        grams.extend(seg[i:i + 2] for i in range(len(seg) - 1))
    return grams


def parse_google_query(q):
    """解析 Google 语法查询：-keyword 排除（仅当 - 前有空格/独立成词），其余为包含。
    标点（- : () 等）前面无空格时视为关键词的一部分，不触发高级搜索语法。"""
    if not q:
        return [], []
    q = normalize_query(q)
    parts = q.strip().split()
    inc, exc = [], []
    for p in parts:
        if p.startswith("-") and len(p) > 1:
            exc.append(p[1:])
        else:
            inc.append(p)
    return inc, exc



# ─── Gov DB 部分 ───
def normalize_query(q):
    """Replace punctuation (Chinese + English) with spaces.
    保留 - : ( ) （ ） 作为关键词一部分（前面无空格时视为普通字符，避免 FTS 高级语法误触发）。"""
    if not q:
        return q
    punct_set = set(string.punctuation.replace("-", "").replace(":", "").replace("(", "").replace(")", "")
                    + "，。、；？！…—·～「」『』〖〗【】〔〕〈〉《》")
    result = "".join(" " if c in punct_set else c for c in q)
    result = re.sub(r"\s+", " ", result).strip()
    return result


def sanitize_fts_query(q):
    """Remove FTS5 special chars and replace punctuation with spaces.
    FTS5 特殊字符: ( ) * ^ \" [ ] . / - : 都替换为空格，避免语法错误（no such column / syntax error）。"""
    if not q:
        return q
    q = normalize_query(q)
    q = re.sub(r'[()（）*^"\[\]./:\-]', ' ', q)
    q = re.sub(r'\s+', ' ', q).strip()
    return q



def get_db():

    conn = sqlite3.connect(DB_PATH, timeout=10)

    conn.row_factory = sqlite3.Row
    conn.execute("PRAGMA busy_timeout=5000")

    conn.execute('PRAGMA journal_mode=WAL')

    return conn



# ===== originals 标题匹配（项目ID展示）=====

_ORIGINALS_DB = '/root/originals.db'

_ORIGINALS_IDX = None          # (proj_titles, comp_titles) 各为 [(title, id_code), ...]
_ORIGINALS_IDX_MT = 0.0        # 索引对应的 originals.db mtime


def _clean_otitle(t):
    """清洗 originals 标题: 去 tab 前缀 / 数字前缀 / 尾标点"""
    if not t:
        return ''
    t = t.strip()
    if '\t' in t:
        t = t.split('\t')[-1].strip()
    t = re.sub(r'^\d{6,10}\s+', '', t)
    t = re.sub(r'[，。、；：\s]+$', '', t)
    return t


_OPROJ_KEY = re.compile(r'(项目|工程|装置|年产|万吨|扩建|技改|新建|生产线|车间)')


def _load_originals_index():
    """懒加载 originals.db 标题索引, mtime 变化时重建. 返回 (proj_titles, comp_titles)"""
    global _ORIGINALS_IDX, _ORIGINALS_IDX_MT
    try:
        mt = os.path.getmtime(_ORIGINALS_DB)
    except OSError:
        return _ORIGINALS_IDX or ([], [])
    if _ORIGINALS_IDX is not None and mt == _ORIGINALS_IDX_MT:
        return _ORIGINALS_IDX
    proj, comp = [], []
    seen_p, seen_c = set(), set()
    if os.path.exists(_ORIGINALS_DB):
        try:
            conn = sqlite3.connect(_ORIGINALS_DB, timeout=5)
            rows = conn.execute('SELECT id_code, title FROM tsk_data').fetchall()
            conn.close()
            for code, t in rows:
                ct = _clean_otitle(t)
                if len(ct) < 8:
                    continue
                if len(ct) >= 12 and _OPROJ_KEY.search(ct):
                    if ct not in seen_p:
                        seen_p.add(ct)
                        proj.append((ct, code))
                elif 8 <= len(ct) <= 30:
                    if ct not in seen_c:
                        seen_c.add(ct)
                        comp.append((ct, code))
        except Exception:
            pass
    # 长标题优先: 匹配到最具体的那条
    proj.sort(key=lambda x: len(x[0]), reverse=True)
    comp.sort(key=lambda x: len(x[0]), reverse=True)
    _ORIGINALS_IDX = (proj, comp)
    _ORIGINALS_IDX_MT = mt
    return _ORIGINALS_IDX


def find_originals_id(title):
    """按标题匹配 originals 项目 → (id_code, level), level: 'proj'|'comp'|None"""
    if not title:
        return None, None
    st = title.strip()
    st = re.sub(r'(环境影响报告书|环评|环境影响评价|公示|受理|审批|批复|决定).*$', '', st)
    st = re.sub(r'[，。、；：\s]+$', '', st)
    if len(st) < 10:
        return None, None
    proj, comp = _load_originals_index()
    for octitle, code in proj:
        if len(octitle) > len(st):
            continue
        if octitle in st:
            return code, 'proj'
    for octitle, code in comp:
        if len(octitle) > len(st):
            continue
        if octitle in st:
            return code, 'comp'
    return None, None


def _orig_id_label(id_code):
    """9位=Project ID, 7位=Plant ID, 其余=None"""
    s = str(id_code) if id_code is not None else ''
    if re.fullmatch(r'\d{9}', s):
        return '项目ID', 'proj'
    if re.fullmatch(r'\d{7}', s):
        return '企业ID', 'plant'
    return None, None



def search(query, page=1, page_size=10, sort="relevance", phase=None, category=None, daterange="all", industry=None):
    """返回 (rows, total, has_more)"""
    if len(str(query).strip().replace(' ', '')) <= 2:
        return [], 0, False  # 2026-09-03 短词守卫: 不执行检索
    conn = sqlite3.connect(DB_PATH, timeout=10)
    conn.row_factory = sqlite3.Row
    conn.execute('PRAGMA journal_mode=WAL')
    c = conn.cursor()
    params = []; where_parts = []
    use_fts = False
    if query:
        use_fts = True
        params.append(query)
    if phase:
        kw = PHASE_MAP[phase]['keywords']
        if not use_fts:
            use_fts = True
            params.append(kw)
        else:
            query += ' AND (' + kw + ')'
            params[0] = query
    if category:
        where_parts.append("r.category = ?"); params.append(category)
    if industry and industry != 'all':
        where_parts.append("r.industry = ?"); params.append(industry)
    if daterange and daterange != 'all':
        days = {'1d':1,'7d':7,'30d':30}[daterange]
        # 格式防御: 非标准 YYYY-MM-DD 日期(中文格式/空/乱码)不进时间窗口, 避免字符串比较误判污染
        where_parts.append(f"(r.publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' AND SUBSTR(r.publish_date,1,10) >= date('now', '-{days} days'))")
    order_sql = 'rank' if sort == 'relevance' else 'r.publish_date DESC' if sort == 'date_desc' else 'r.publish_date ASC'
    fetch_size = page_size + 1
    offset = (page-1)*page_size
    total = 0
    if use_fts:
        q = params[0]
        # 标点符号统一视为空格
        q = normalize_query(q)
        params[0] = q
        # 解析 Google 语法：空格=AND, -keyword=排除
        inc_terms, exc_terms = parse_google_query(q)
        # ── gov_entity 精确实体反查 (2026-09-03): 完整公司名/电话 出现在正文也命中 ──
        if not exc_terms and len(inc_terms) == 1 and ' ' not in q.strip():
            _term = inc_terms[0]
            _ent = None
            if contains_chinese(_term) and len(_term) >= 3:
                _ent = ('company', _term)
            else:
                _dig = re.sub(r'\D', '', _term)
                if 7 <= len(_dig) <= 13:
                    _ent = ('phone', _dig)
            if _ent:
                _ew = ' AND ' + ' AND '.join(where_parts) if where_parts else ''
                try:
                    c.execute(f'SELECT COUNT(*) FROM gov_entity e JOIN gov_raw r ON r.id = e.doc_id WHERE e.etype = ? AND e.value = ?{_ew}', (_ent[0], _ent[1]) + tuple(params[1:]))
                    _et = c.fetchone()[0]
                    if _et > 0:
                        c.execute(f'SELECT r.*, "" AS rank FROM gov_entity e JOIN gov_raw r ON r.id = e.doc_id WHERE e.etype = ? AND e.value = ?{_ew} ORDER BY r.publish_date DESC, r.id DESC LIMIT ? OFFSET ?', (_ent[0], _ent[1]) + tuple(params[1:]) + (fetch_size, offset))
                        rows = [dict(r) for r in c.fetchall()]
                        has_more = len(rows) > page_size
                        if has_more: rows = rows[:page_size]
                        conn.close()
                        return rows, _et, has_more
                except sqlite3.OperationalError:
                    pass  # 实体表异常 → 回落常规搜索
        # 单英文词(域名/URL) → 走 page_url 索引快路径 (LIKE 'http://%kw%' 常量前缀命中 idx_gov_raw_page_url)
        if len(inc_terms) == 1 and not contains_chinese(q) and not exc_terms:
            kw = inc_terms[0]
            # 关键: SQLite LIKE 索引优化只对编译期字面量生效; 参数化 ? 会退化为全表扫(~43s)
            # 字面量 'http://%kw%' 走 idx_gov_raw_page_url 范围扫描 ~1.5s。
            # kw 必须白名单校验(域名/URL 只含字母数字._-), 防注入
            if re.fullmatch(r'[A-Za-z0-9._\-]+', kw):
                url_pat = 'http://%' + kw + '%'
                url_pat2 = 'https://%' + kw + '%'
                fts_where_sql = ''
                if where_parts:
                    fts_where_sql = ' AND ' + ' AND '.join(where_parts)
                try:
                    c.execute(f"SELECT COUNT(*) FROM gov_raw r WHERE (r.page_url LIKE '{url_pat}' OR r.page_url LIKE '{url_pat2}'){fts_where_sql}",
                              tuple(params[1:]))
                    total = c.fetchone()[0]
                    if total > 0:
                        # 三步: ①只取 rowid (idx_gov_raw_page_url 覆盖索引 1.5s)
                        #       ②批量 IN 查 publish_date (主键 0.1s)
                        #       ③Python 排序分页 + 主键取数 (0.0s)
                        # 直接 SELECT 列 / 子查询 ORDER BY 都会逐行回表 ~20s
                        c.execute(f"SELECT r.rowid FROM gov_raw r WHERE (r.page_url LIKE '{url_pat}' OR r.page_url LIKE '{url_pat2}'){fts_where_sql}",
                                  tuple(params[1:]))
                        ids = [r[0] for r in c.fetchall()]
                        if not ids:
                            conn.close()
                            return [], 0, False
                        ph_ids = ','.join('?' * len(ids))
                        c.execute(f'SELECT id, publish_date FROM gov_raw r WHERE r.id IN ({ph_ids})', ids)
                        id_dates = c.fetchall()
                        # 按日期倒序 (空日期排最后), 稳定分页
                        id_dates.sort(key=lambda x: x[1] or '', reverse=True)
                        page_ids = [rid for rid, _ in id_dates[offset:offset + fetch_size]]
                        has_more = len(page_ids) > page_size
                        if has_more: page_ids = page_ids[:page_size]
                        if not page_ids:
                            rows = []
                        else:
                            ph2 = ','.join('?' * len(page_ids))
                            c.execute(f'SELECT * FROM gov_raw r WHERE r.id IN ({ph2})', page_ids)
                            rows = [dict(r) for r in c.fetchall()]
                        conn.close()
                        return rows, total, has_more
                    else:
                        # page_url 无命中 → 单英文词(域名/日期/编号)直接返回空，
                        # 避免落 FTS 表 LIKE 全扫卡死 (如 '2026-08-24' 0 命中时)
                        conn.close()
                        return [], 0, False
                except Exception:
                    pass
            # page_url 无命中时兜底 title/site_name LIKE (可能慢但仅限异常)
        # 中文 / 多词 / 排除词 / 单英文词(域名URL) → LIKE 逐词匹配
        if contains_chinese(q) or exc_terms or len(inc_terms) > 1 or (len(inc_terms) == 1 and not contains_chinese(q)):
            # 纯中文多词(每词≥3字, 空格=AND) → trigram FTS 优先; 排除词用 NOT LIKE 二次过滤
            # 0命中不直接空返 → 继续落 bigram/轻量LIKE 兜底 (bigram 2026-09-03 已恢复)
            if contains_chinese(q) and all(len(t) >= 3 for t in inc_terms + exc_terms):
                try:
                    _fts_q = sanitize_fts_query(' '.join(inc_terms)) if inc_terms else ''
                    _w_list = ['gov_search MATCH ?']
                    _fparams = [_fts_q]
                    if where_parts:
                        _w_list.extend(where_parts)
                        _fparams.extend(params[1:])
                    for _t in exc_terms:
                        _w_list.append('(r.title NOT LIKE ? AND r.site_name NOT LIKE ? AND r.page_url NOT LIKE ?)')
                        _fparams.extend(['%' + _t + '%'] * 3)
                    _w_all = ' AND '.join(_w_list)
                    c.execute(f'SELECT COUNT(*) FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE {_w_all}', _fparams)
                    total = c.fetchone()[0]
                    if total > 0:
                        c.execute(f'SELECT r.*, rank FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE {_w_all} ORDER BY {order_sql} LIMIT ? OFFSET ?', _fparams + [fetch_size, offset])
                        rows = [dict(r) for r in c.fetchall()]
                        has_more = len(rows) > page_size
                        if has_more: rows = rows[:page_size]
                        conn.close()
                        return rows, total, has_more
                except sqlite3.OperationalError:
                    pass
            like_conds = []
            like_params = []
            for t in inc_terms:
                like_conds.append('(r.title LIKE ? OR r.site_name LIKE ? OR r.page_url LIKE ?)')
                like_params.extend(['%' + t + '%', '%' + t + '%', '%' + t + '%'])
            for t in exc_terms:
                like_conds.append('(r.title NOT LIKE ? AND r.site_name NOT LIKE ? AND r.page_url NOT LIKE ?)')
                like_params.extend(['%' + t + '%', '%' + t + '%', '%' + t + '%'])
            if not like_conds:
                rows = []; total = 0; has_more = False
                conn.close()
                return rows, total, has_more
            # 中文词(≥2字) → bigram 索引快路径 (2026-09-03 恢复; LIKE %xx% 全扫 45s, gov_bigram 索引 0.01s)
            # 覆盖: 2字词直接 gram; ≥3字词拆 2-gram 序列交集 (FTS trigram 对错序词 0 命中时不再落 LIKE 全扫)
            # 含标点 token (项目-招标 / 环境:保护) 用 token_grams 提取中文段 gram, 不再触发 FTS 语法错误
            if (inc_terms
                    and all(len(t) >= 2 and contains_chinese(t) and token_grams(t) for t in inc_terms)):
                try:
                    # 交集 rowid 集合 (纯 bigram 索引 COUNT, 不回表 ~0.01s)
                    # ≥3字词: 拆 2-gram 序列 (醋酸乙烯 -> 醋酸/酸乙/乙烯), 各 gram 交集 ≈ 包含该词
                    id_sets = []
                    for t in inc_terms:
                        grams = token_grams(t)
                        term_ids = None
                        for g in grams:
                            c.execute('SELECT rowid FROM gov_bigram WHERE gram = ?', (g,))
                            s = set(r[0] for r in c.fetchall())
                            if term_ids is None:
                                term_ids = s
                            else:
                                term_ids &= s
                            if not term_ids:
                                break
                        id_sets.append(term_ids or set())
                    ids = set.intersection(*id_sets) if id_sets else set()
                    if not ids:
                        # 交集为空 → 直接返回空 (不再落 LIKE 全扫 45s)
                        conn.close()
                        return [], 0, False
                    if ids:
                        id_list = list(ids)
                        # 排除词 → bigram 集合差 (title/site_name 不含该词 ≈ gram 集合差)
                        # NOT LIKE 全表扫 37s / EXISTS COUNT 37s / 分批 IN 26s 全卡死；
                        # bigram 集合差 0.03s。page_url 排除差异极少见，可接受。
                        for t in exc_terms:
                            exc_grams = token_grams(t)
                            if not exc_grams:
                                continue
                            for g in exc_grams:
                                c.execute('SELECT rowid FROM gov_bigram WHERE gram = ?', (g,))
                                exc_s = set(r[0] for r in c.fetchall())
                                ids = ids - exc_s
                        id_list = list(ids)
                        if not id_list:
                            conn.close()
                            return [], 0, False
                        # 带日期/行业过滤时需回表精确计数; 否则用 bigram 集合大小
                        extra_join = ' AND ' + ' AND '.join(where_parts) if where_parts else ''
                        if where_parts:
                            # 大集合(如"项目"26万)时禁止 WHERE id IN (N个参数) —— SQLite 999 变量上限会抛异常
                            # 被 except 吞掉后落 FTS 表 LIKE 全扫卡死。改用 EXISTS + 过滤条件直接 COUNT (流式, 走日期索引)
                            gram_counts = [len(token_grams(t)) for t in inc_terms]
                            ex_conds = ' AND '.join(
                                ' AND '.join(f'EXISTS (SELECT 1 FROM gov_bigram b{i}_{j} WHERE b{i}_{j}.rowid=r.id AND b{i}_{j}.gram=?)'
                                             for j in range(gram_counts[i]))
                                for i in range(len(inc_terms)))
                            ex_params = []
                            for t in inc_terms:
                                ex_params.extend(token_grams(t))
                            count_sql = f'SELECT COUNT(*) FROM gov_raw r WHERE {ex_conds}{extra_join}'
                            c.execute(count_sql, ex_params)
                            total = c.fetchone()[0]
                        else:
                            total = len(id_list)
                        if total == 0:
                            conn.close()
                            return [], 0, False
                        # 分页: 从 gov_raw 按 publish_date 索引倒序扫描 + EXISTS 校验 (0.00s, 找到即停)
                        # 避免大词(公示26万)全集合排序; 日期索引保证流式取到当页即止
                        # ≥3字词同样拆 2-gram, 每个 gram 都须 EXISTS (AND 语义)
                        gram_counts = [len(token_grams(t)) for t in inc_terms]
                        ex_conds = ' AND '.join(
                            ' AND '.join(f'EXISTS (SELECT 1 FROM gov_bigram b{i}_{j} WHERE b{i}_{j}.rowid=r.id AND b{i}_{j}.gram=?)'
                                         for j in range(gram_counts[i]))
                            for i in range(len(inc_terms)))
                        ex_params = []
                        for t in inc_terms:
                            ex_params.extend(token_grams(t))
                        if where_parts:
                            where_sql_join = extra_join
                        else:
                            where_sql_join = ''
                        # 排除词 → 分页同样排除 (多取3倍再按剩余集合过滤, 保证页满; 排除比例高的极端场景分页会略少, 但绝不卡死)
                        if exc_terms and id_list:
                            limit_sql = f'SELECT r.id FROM gov_raw r WHERE {ex_conds}{where_sql_join} ORDER BY r.publish_date DESC LIMIT ? OFFSET ?'
                            c.execute(limit_sql, ex_params + [fetch_size * 3, offset])
                            page_ids = [r[0] for r in c.fetchall()]
                            page_ids = [pid for pid in page_ids if pid in ids]
                        else:
                            limit_sql = f'SELECT r.id FROM gov_raw r WHERE {ex_conds}{where_sql_join} ORDER BY r.publish_date DESC LIMIT ? OFFSET ?'
                            c.execute(limit_sql, ex_params + [fetch_size, offset])
                            page_ids = [r[0] for r in c.fetchall()]
                        has_more = len(page_ids) > page_size
                        if has_more: page_ids = page_ids[:page_size]
                        if page_ids:
                            ph2 = ','.join('?' * len(page_ids))
                            c.execute(f'SELECT * FROM gov_raw r WHERE r.id IN ({ph2})', page_ids)
                            rows = [dict(r) for r in c.fetchall()]
                        else:
                            rows = []
                        conn.close()
                        return rows, total, has_more
                except Exception:
                    pass
            # Use FTS table for LIKE (fast, ~382K rows) JOIN gov_raw (rowid 已对齐)
            fts_conds = []
            for t in inc_terms:
                fts_conds.append('(s.title LIKE ? OR s.site_name LIKE ? OR r.page_url LIKE ?)')
            for t in exc_terms:
                fts_conds.append('(s.title NOT LIKE ? AND s.site_name NOT LIKE ? AND r.page_url NOT LIKE ?)')
            fts_where = ' AND '.join(fts_conds)
            fts_params = list(like_params) + params[1:]
            if where_parts:
                raw_where = ' AND '.join(where_parts)
                fts_where += ' AND ' + raw_where
            c.execute(f'SELECT COUNT(*) FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE {fts_where}', fts_params)
            total = c.fetchone()[0]
            c.execute(f'SELECT r.*, "" as rank FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE {fts_where} ORDER BY r.publish_date DESC LIMIT ? OFFSET ?', fts_params + [fetch_size, offset])
            rows = [dict(r) for r in c.fetchall()]
            has_more = (offset + len(rows)) < total
            if has_more:
                rows = rows[:page_size]
            conn.close()
            return rows, total, has_more
        else:
            fts_where_sql = ''
            if where_parts:
                fts_where_sql = ' AND ' + ' AND '.join(where_parts)
            try:
                c.execute(f'SELECT COUNT(*) FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE gov_search MATCH ?{fts_where_sql}', (sanitize_fts_query(q),) + tuple(params[1:]))
                total = c.fetchone()[0]
                c.execute(f'SELECT r.*, rank FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE gov_search MATCH ?{fts_where_sql} ORDER BY {order_sql} LIMIT ? OFFSET ?', (sanitize_fts_query(q),) + tuple(params[1:]) + (fetch_size, offset))
            except sqlite3.OperationalError:
                like_q = '%' + q + '%'
                c.execute(f'SELECT COUNT(*) FROM gov_raw WHERE (title LIKE ? OR summary LIKE ?){(" AND " + " AND ".join(where_parts)) if where_parts else ""}', (like_q, like_q) + tuple(params[1:]))
                total = c.fetchone()[0]
                c.execute(f'SELECT r.*, "" as rank FROM gov_raw r WHERE (title LIKE ? OR summary LIKE ?){(" AND " + " AND ".join(where_parts)) if where_parts else ""} ORDER BY r.publish_date DESC LIMIT ? OFFSET ?', (like_q, like_q) + tuple(params[1:]) + (fetch_size, offset))
    else:
        where = 'WHERE ' + ' AND '.join(where_parts) if where_parts else ''
        c.execute(f'SELECT COUNT(*) FROM gov_raw r {where}', params)
        total = c.fetchone()[0]
        c.execute(f'SELECT r.* FROM gov_raw r {where} ORDER BY {order_sql} LIMIT ? OFFSET ?', params + [fetch_size, offset])
    rows = [dict(r) for r in c.fetchall()]
    has_more = len(rows) > page_size
    if has_more:
        rows = rows[:page_size]
    conn.close()
    return rows, total, has_more

def get_detail(record_id):

    conn = get_db(); c = conn.cursor()

    c.execute("SELECT * FROM gov_raw WHERE id=?", (record_id,))

    r = c.fetchone()

    conn.close()

    return dict(r) if r else None



# ─── CCEUP/中项网 DB ───

def cceup_search(q, page=1, per=20, daterange='all', industry=None):

    if not os.path.exists(CONTACT_DB_PATH): return 0, []

    db = sqlite3.connect(CONTACT_DB_PATH); db.row_factory = sqlite3.Row

    conds, params = [], []

    if q:

        inc, exc = parse_google_query(q)

        for t in inc:

            conds.append('(project_id LIKE ? OR project_name LIKE ? OR owner_company LIKE ? OR industry LIKE ? OR province LIKE ? OR city LIKE ?)')

            params.extend(['%'+t+'%']*6)

        for t in exc:

            conds.append('(project_id NOT LIKE ? AND project_name NOT LIKE ? AND owner_company NOT LIKE ? AND industry NOT LIKE ? AND province NOT LIKE ? AND city NOT LIKE ?)')

            params.extend(['%'+t+'%']*6)

    if industry and industry != 'all' and industry in INDUSTRY_MAP:

        kws = INDUSTRY_MAP[industry]['kw'][:4]

        if kws:

            sub_conds = ' OR '.join(['industry LIKE ?'] * len(kws))

            conds.append('(' + sub_conds + ')')

            params.extend(['%'+k+'%' for k in kws])

    if daterange and daterange != 'all':

        days = {'1d':1,'7d':7,'30d':30}[daterange]

        conds.append(f"(publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' AND SUBSTR(publish_date,1,10) >= date('now', '-{days} days'))")

    wh = 'WHERE '+' AND '.join(conds) if conds else ''

    total = db.execute('SELECT COUNT(*) FROM cceup_projects '+wh, params).fetchone()[0]

    rows = [dict(r) for r in db.execute('SELECT * FROM cceup_projects '+wh+' ORDER BY publish_date DESC, project_id LIMIT ? OFFSET ?', params+[per,(page-1)*per])]

    db.close()

    return total, rows



def cceup_contacts_count(pid):

    if not os.path.exists(CONTACT_DB_PATH): return 0

    db = sqlite3.connect(CONTACT_DB_PATH)

    total = db.execute('SELECT COUNT(*) FROM cceup_contacts WHERE project_id=?', (pid,)).fetchone()[0]

    db.close()

    return total



# ─── ZNLH/中能联合 DB ───

def znlh_search(q, page=1, per=20, daterange='all', industry=None):

    if not os.path.exists(ZNLH_DB_PATH): return 0, []

    db = sqlite3.connect(ZNLH_DB_PATH); db.row_factory = sqlite3.Row

    conds, params = [], []

    if q:

        inc, exc = parse_google_query(q)

        for t in inc:

            conds.append('(project_id LIKE ? OR project_name LIKE ? OR owner_company LIKE ? OR industry LIKE ? OR province LIKE ? OR city LIKE ?)')

            params.extend(['%'+t+'%']*6)

        for t in exc:

            conds.append('(project_id NOT LIKE ? AND project_name NOT LIKE ? AND owner_company NOT LIKE ? AND industry NOT LIKE ? AND province NOT LIKE ? AND city NOT LIKE ?)')

            params.extend(['%'+t+'%']*6)

    if industry and industry != 'all' and industry in INDUSTRY_MAP:

        kws = INDUSTRY_MAP[industry]['kw'][:4]

        if kws:

            sub_conds = ' OR '.join(['industry LIKE ?'] * len(kws))

            conds.append('(' + sub_conds + ')')

            params.extend(['%'+k+'%' for k in kws])

    if daterange and daterange != 'all':

        days = {'1d':1,'7d':7,'30d':30}[daterange]

        conds.append(f"(publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' AND SUBSTR(publish_date,1,10) >= date('now', '-{days} days'))")

    wh = 'WHERE '+' AND '.join(conds) if conds else ''

    total = db.execute('SELECT COUNT(*) FROM znlh_projects '+wh, params).fetchone()[0]

    rows = [dict(r) for r in db.execute('SELECT * FROM znlh_projects '+wh+' ORDER BY publish_date DESC, project_id LIMIT ? OFFSET ?', params+[per,(page-1)*per])]

    db.close()

    return total, rows



def znlh_detail(pid):

    if not os.path.exists(ZNLH_DB_PATH): return None

    db = sqlite3.connect(ZNLH_DB_PATH); db.row_factory = sqlite3.Row

    r = db.execute('SELECT * FROM znlh_projects WHERE project_id=?',(pid,)).fetchone()

    if not r: db.close(); return None

    p = dict(r)

    p['contacts'] = [dict(r) for r in db.execute('SELECT * FROM znlh_contacts WHERE project_id=? ORDER BY id',(pid,))]

    db.close(); return p



# ─── ZC/中策大数据 DB ───

def zc_search(q, page=1, per=20, daterange='all', industry=None):

    if not os.path.exists(ZC_DB_PATH): return 0, []

    db = sqlite3.connect(ZC_DB_PATH); db.row_factory = sqlite3.Row

    conds, params = [], []

    if q:

        inc, exc = parse_google_query(q)

        for t in inc:

            conds.append('(project_id LIKE ? OR project_name LIKE ? OR industry LIKE ? OR province LIKE ? OR city LIKE ?)')

            params.extend(['%'+t+'%']*5)

        for t in exc:

            conds.append('(project_id NOT LIKE ? AND project_name NOT LIKE ? AND industry NOT LIKE ? AND province NOT LIKE ? AND city NOT LIKE ?)')

            params.extend(['%'+t+'%']*5)

    if industry and industry != 'all' and industry in INDUSTRY_MAP:

        kws = INDUSTRY_MAP[industry]['kw'][:4]

        if kws:

            sub_conds = ' OR '.join(['industry LIKE ?'] * len(kws))

            conds.append('(' + sub_conds + ')')

            params.extend(['%'+k+'%' for k in kws])

    if daterange and daterange != 'all':

        days = {'1d':1,'7d':7,'30d':30}[daterange]

        conds.append(f"(publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' AND SUBSTR(publish_date,1,10) >= date('now', '-{days} days'))")

    wh = 'WHERE '+' AND '.join(conds) if conds else ''

    total = db.execute('SELECT COUNT(*) FROM zc_projects '+wh, params).fetchone()[0]

    rows = [dict(r) for r in db.execute('SELECT * FROM zc_projects '+wh+' ORDER BY publish_date DESC, project_id LIMIT ? OFFSET ?', params+[per,(page-1)*per])]

    db.close()

    return total, rows



def zc_detail(pid):

    if not os.path.exists(ZC_DB_PATH): return None

    db = sqlite3.connect(ZC_DB_PATH); db.row_factory = sqlite3.Row

    r = db.execute('SELECT * FROM zc_projects WHERE project_id=?',(pid,)).fetchone()

    if not r: db.close(); return None

    p = dict(r)

    p['contacts'] = [dict(r) for r in db.execute('SELECT * FROM zc_contacts WHERE project_id=? ORDER BY id',(pid,))]

    db.close(); return p





# ══════════════════════════════════════════════

#  HTTP Handler

# ══════════════════════════════════════════════



# ─── Crawler DB ───

def crawler_detail(pid):

    if not os.path.exists(DB_PATH): return None

    db = sqlite3.connect(DB_PATH); db.row_factory = sqlite3.Row

    try:

        r = db.execute('SELECT * FROM gov_raw WHERE id=?',(pid,)).fetchone()

        if not r: return None

        d = dict(r)

        d['url'] = d.get('page_url') or d.get('source_url') or ''

        d['domain'] = d.get('site_name') or ''

        return d

    finally:

        db.close()





# ─── Project visit tracking helpers ───

def _track_project_visit(ptype, pid, username):

    if not username or not pid: return

    db = sqlite3.connect(SESSION_DB)

    try:

        db.execute('INSERT INTO project_visits (project_type, project_id, username) VALUES (?,?,?)',

                   (ptype, str(pid), username))

        db.commit()

    except:

        pass

    finally:

        db.close()



def _get_project_visitors(ptype, pid, limit=3):

    if not pid: return 0, []

    db = sqlite3.connect(SESSION_DB)

    db.row_factory = sqlite3.Row

    try:

        total = db.execute('SELECT COUNT(*) FROM project_visits WHERE project_type=? AND project_id=?',

                           (ptype, str(pid))).fetchone()[0]

        recent = [dict(r) for r in db.execute(

            'SELECT DISTINCT username, visited_at FROM project_visits WHERE project_type=? AND project_id=? ORDER BY visited_at DESC LIMIT ?',

            (ptype, str(pid), limit))]

        unique_count = db.execute('SELECT COUNT(DISTINCT username) FROM project_visits WHERE project_type=? AND project_id=?',

                                   (ptype, str(pid))).fetchone()[0]

        return unique_count, recent

    except:

        return 0, []

    finally:

        db.close()



# ─── Login / Session ───

ALLOWED_USERS = {'wjianning', 'lzhong', 'steveou', 'schengbo', 'admin', 'bjiir'}


NAV_CLUSTER_CSS = (
    '<style>'
    '.nav-right{display:flex;align-items:center;gap:10px;margin-left:auto;font-size:13px;color:#5f6368}'
    '.nav-right.fixed{position:fixed;top:12px;right:20px;z-index:900;margin-left:0}'
    '@media(min-width:1760px){.nav-right{position:fixed;top:12px;right:20px;z-index:900;margin-left:0}}'
    '.nav-right a.nav-admin,.nav-right a.nav-tm{display:inline-flex;align-items:center;'
    'padding:6px 14px;border-radius:999px;font-size:13px;font-weight:600;cursor:pointer;'
    'text-decoration:none;line-height:1.2;transition:background .15s,border-color .15s,color .15s}'
    '.nav-right a.nav-admin{background:#1a73e8;color:#fff;border:1px solid #1a73e8;'
    'box-shadow:0 1px 4px rgba(26,115,232,.3)}'
    '.nav-right a.nav-admin:hover{background:#1557b0;border-color:#1557b0;color:#fff;text-decoration:none}'
    '.nav-right a.nav-tm{background:#fff;color:#1a73e8;border:1px solid #c3d7fb}'
    '.nav-right a.nav-tm:hover{background:#e8f0fe;border-color:#1a73e8;text-decoration:none}'
    '.nav-right .nav-user{font-size:12px;color:#5f6368}'
    '.nav-right a.nav-logout{display:inline-flex;align-items:center;padding:6px 12px;border-radius:999px;'
    'font-size:13px;color:#d93025;background:#fff;border:1px solid #f0c4bf;text-decoration:none;'
    'transition:background .15s}'
    '.nav-right a.nav-logout:hover{background:#fdecea;text-decoration:none}'
    # 2026-09-11 移动端：固定定位会撑破窄屏（实测 375px 溢出 6px）→ 窄屏改静态并允许换行
    '@media(max-width:760px){.nav-right{position:static !important;margin-left:0;width:100%;'
    'flex-wrap:wrap;gap:8px;justify-content:flex-start}}'
    '</style>'
)

# 统一分页条样式（2026-09-10）—— 自包含，paginate() 与 admin 脚本表共用。
# 设计要点（用户反馈「底部的页码与分页按钮太靠底部了，不容易看清楚」）：
# ① 白色卡片 + 边框 + 投影 → 与页面背景分离，醒目
# ② margin-bottom 60px → 不再贴着窗口底边
# ③ 38px 高按钮 + hover/active 态 → 好点、易看清
PAGER_CSS = (
    '<style>'
    '.pagination{display:flex;align-items:center;justify-content:center;gap:6px;flex-wrap:wrap;'
    'margin:26px 0 60px;padding:14px 18px;background:#fff;border:1px solid #eceff3;border-radius:12px;'
    'box-shadow:0 1px 3px rgba(0,0,0,.06)}'
    '.pagination .page-btn{display:inline-flex;align-items:center;justify-content:center;min-width:38px;height:38px;'
    'padding:0 12px;border:1px solid #dadce0;border-radius:8px;background:#fff;color:#1a73e8;font-size:13px;'
    'line-height:1;text-decoration:none;transition:background .15s,border-color .15s,color .15s}'
    '.pagination .page-btn:hover{background:#e8f0fe;border-color:#1a73e8;color:#1a73e8;text-decoration:none}'
    '.pagination .page-btn.active{background:#1a73e8;color:#fff;border-color:#1a73e8;font-weight:600}'
    '.pagination .page-ellipsis{color:#9aa0a6;padding:0 4px;font-size:13px}'
    '.pagination .page-info{font-size:13px;color:#5f6368;margin-right:10px}'
    '.pagination .page-info b{color:#1a73e8}'
    '.pagination .page-jump{display:inline-flex;align-items:center;gap:4px;font-size:13px;color:#5f6368}'
    '.pagination .page-jump input{width:54px;padding:7px 8px;border:1px solid #dadce0;border-radius:8px;'
    'font-size:13px;text-align:center;outline:none}'
    '.pagination .page-jump input:focus{border-color:#1a73e8}'
    '</style>'
)


def nav_right_html(username, fixed=False):
    """右上角导航簇（2026-09-10 用户需求；样式自包含，不依赖页面 CSS）。

    - 「登录用户」从主页面居中位置移到右上角，并新增「退出」按钮（主页面此前无退出）
    - 「TM Script Center」→ /tm（反代 127.0.0.1:8010）；「Admin」→ /crawler/admin
    - Admin 可见性：**除 bjiir 外所有登录用户可见**（bjiir 看不到；用户 2026-09-10 确认）
    - 工具市场可见性：**同样对 bjiir 隐藏**（2026-09-11 用户需求；与 Admin 同一口径 = 入口级隐藏，
      路由不做角色校验，与 Admin 现状一致）
    - 顺位：Admin · 工具市场 · TM Script Center · 登录用户 · 退出（Admin 在 TM 入口左侧）
    - fixed=True → 常驻浏览器窗口右上角（主页面用）；否则在流内右对齐，
      且仅在视口 ≥1760px 时脱离文档流贴窗口右上角（避免与页内 tab 行重叠）
    """
    u = username or ''
    parts = []
    if u and u != 'bjiir':
        parts.append('<a class="nav-admin" href="/crawler/admin">Admin</a>')
        parts.append('<a class="nav-tm" href="/tools">工具市场</a>')
    parts.append('<a class="nav-tm" href="/tm/">TM Script Center</a>')
    parts.append('<span class="nav-user">' + u + '</span>')
    parts.append('<a class="nav-logout" href="/logout">退出</a>')
    cls = 'nav-right fixed' if fixed else 'nav-right'
    return NAV_CLUSTER_CSS + '<div class="' + cls + '">' + ''.join(parts) + '</div>'

SESSION_DB = '/root/session.db'  # 2026-09-10: 会话/访问统计迁出 contact.db(contact.db 变纯联系人内容库)



def init_session_db():

    db = sqlite3.connect(SESSION_DB)

    db.execute('''CREATE TABLE IF NOT EXISTS sessions (

        token TEXT PRIMARY KEY, username TEXT NOT NULL,

        created_at TEXT DEFAULT (datetime('now','localtime')),

        last_seen TEXT DEFAULT (datetime('now','localtime'))

    )''')

    db.execute('''CREATE TABLE IF NOT EXISTS usage_log (

        id INTEGER PRIMARY KEY AUTOINCREMENT, username TEXT NOT NULL,

        action TEXT NOT NULL, page TEXT DEFAULT '', query TEXT DEFAULT '',

        ip TEXT DEFAULT '', created_at TEXT DEFAULT (datetime('now','localtime'))

    )''')

    db.execute('''CREATE TABLE IF NOT EXISTS project_visits (

        id INTEGER PRIMARY KEY AUTOINCREMENT,

        project_type TEXT NOT NULL,

        project_id TEXT NOT NULL,

        username TEXT NOT NULL,

        visited_at TEXT DEFAULT (datetime('now','localtime'))

    )''')

    db.execute('''CREATE INDEX IF NOT EXISTS idx_pv_lookup 

        ON project_visits(project_type, project_id, visited_at DESC)''')

    db.commit(); db.close()



init_session_db()
ensure_industry_column()



# ─── HTTP Handler ───



# 正文是不是"真 HTML"。
# ⚠️ 绝对不能用 `'<' in content` 判断 —— 正文里出现书名号嵌套（《…<…>…》，
#    如《安徽省大气办关于印发<安徽省2020年大气污染防治重点工作任务>的通知》）
#    或数学比较符（COD<50）都会**误判成 HTML**，然后落进 .ct-html（它没有
#    white-space:pre-line）→ 正文里所有 \n\n 被当空白吞掉 → 整篇塌成一坨。
#    2026-09-11 实例：广德市人民政府-环评审批 id=2097613798365819047，5190 字正文
#    里只有 1 个 '<'（就是上面那个书名号嵌套），整篇无分段。
_BODY_TAG_RE = re.compile(r'</?[a-zA-Z][a-zA-Z0-9]*(?:\s[^<>]*)?/?>')


# ─── 正文噪声块剥离（2026-09-11）─────────────────────────────────────
# 政府站常把整段内联 CSS/JS 塞在正文开头。实例（绍兴上虞区政府-环评公示，
# id=2097613798365821229）的 content 开头就是：
#   <div class="article"><div class="bt-note-30"></div>
#   <style type="text/css">\t.ewb-header {\n\t\tbox-sizing: content-box;\n\t} … </style>
# 不剥的话两个后果：
#   ① 列表摘要变成 CSS 文本 —— `_excerpt()` 里 `re.sub(r'<[^>]+>',' ')` 只删标签，
#      标签**之间**的 CSS 文本原样留下，于是摘要成了 `.ewb-header { box-sizing: content-box; }`；
#   ② 详情页把 <style> **原样注入**（.ct-html 里直接是 <style>）→ 全页 CSS 被污染。
# 实测存量 8,513 / 556,959 行 content 含 <style>（1.5%）。
# ⚠️ 截断兜底：列表取的是 `SUBSTR(content,1,400)`，可能正好切在 </style> **之前**
#    → 配对正则匹配不到，必须有「未闭合开标签截到末尾」的兜底，否则 CSS 一截两断。
_NOISE_BLOCK_RE = re.compile(r'<(script|style|noscript|template)\b[^>]*>.*?</\1\s*>', re.I | re.S)
_NOISE_OPEN_RE = re.compile(r'<(?:script|style|noscript|template)\b[^>]*>.*\Z', re.I | re.S)
_NOISE_COMMENT_RE = re.compile(r'<!--.*?-->', re.S)
_NOISE_LINK_RE = re.compile(r'<link\b[^>]*>', re.I)


def strip_body_noise(s):
    """剥掉正文里的 <script>/<style>/<noscript>/<template> 块（连内容）、HTML 注释、<link>。

    不剥会把内联 CSS 当正文显示（列表摘要变 CSS），并把 <style> 注入详情页污染全页样式。
    """
    s = s or ''
    prev = None
    while prev != s:                 # 先按配对删；残留的再按「未闭合」截到末尾
        prev = s
        s = _NOISE_BLOCK_RE.sub(' ', s)
    s = _NOISE_OPEN_RE.sub(' ', s)
    s = _NOISE_COMMENT_RE.sub(' ', s)
    s = _NOISE_LINK_RE.sub(' ', s)
    return s


# 列表摘要开头的页眉元数据 / 字号条（正文里本该被爬虫剥干净，显示端兜底）。
# 实例：`发布时间：2026-09-07 10:28 信息来源：莆田市生态环境局 点击数： 字号： T | T 根据《…》
_EXC_LABELS = r'(?:发布时间|发布日期|更新时间|信息来源|来源|点击数|浏览量|浏览次数|累计次数|访问次数|访问量|字号|字体大小|字体|分享到|分享|打印)'
_EXC_LABEL_RE = re.compile(r'^' + _EXC_LABELS + r'\s*[：:]\s*')
# 不带值的纯噪声标记（社交分享条 / 视力保护色 / 字号「[小 中 大]」）——
# 它们自己就是「页眉垃圾」的证据，计入标签数参与提交判定
_EXC_BARE_RE = re.compile(r'^(?:微博|QQ空间|QQ好友|微信|新浪微博|人人网|豆瓣|视力保护色|关闭|打印本页'
                          r'|[\[]\s*[小中大]\s*[小中大]\s*[小中大]\s*[\]]|【\s*[小中大]\s*[小中大]\s*[小中大]\s*】)'
                          r'\s*[：:]?\s*')
# 面包屑 / 导航前缀（比「标签：值」更外层，直接在摘要最前面）——
# 实例：`导航菜单 南通LNG深度处理工程环境影响评价第一次公示 来源: 洋口港经济开发区 发布时间：…`
# 不先剥掉它，后面的「标题重复 + 页眉」都因为不在串首而躲过判定。
_EXC_LEAD_RE = re.compile(r'^(?:导航菜单|网站导航|当前位置|您的位置|面包屑|首页(?=\s*[>》»]))\s*[>》»:：]?\s*')
# 「XX政府门户网站：www.xxx.gov.cn」
_EXC_SITE_RE = re.compile(r'^\S{0,10}(?:门户网站|政府网站|官网)\s*[：:]\s*\S{0,60}\s*')
_EXC_VAL_RE = re.compile(r'[^\s，。；、：！？《》〈〉（）()〔〕【】“”"\']{1,15}')
_EXC_TIME_RE = re.compile(r'^\d{1,2}:\d{2}(?::\d{2})?\s*')
_EXC_GLYPH_RE = re.compile(r'^(?:[A-Za-z]\s*[|｜]\s*)+[A-Za-z]?\s*')   # 「T | T」字号条
_EXC_WS_RE = re.compile(r'^\s+')
_EXC_BAR_RE = re.compile(r'^[|｜·]\s*')
_EXC_DATE_RE = re.compile(r'\d{4}\s*[-/年.]\s*\d{1,2}\s*[-/月.]\s*\d{1,2}')
# 页眉里的互斥文件/分享图（`4ef66ef5ca044fd9990b91dd41394aac.jpg`）
_EXC_JUNK_RE = re.compile(r'^(?:[0-9a-fA-F]{16,}(?:\.[A-Za-z0-9]{1,5})?|\S+\.(?:jpg|jpeg|png|gif|bmp|webp|pdf|docx?|xlsx?|zip))\s*')


def _exc_follow_ok(s):
    """值的下一个 token 必须是另一个标签 / 时间 / 字号条 / 竖线，或串尾。

    ⚠️ 否则这个「值」其实正文：实测 `分享： 周佳、鲍佳萍因保管不善…`，
    不加这道闸就会把真人名「周佳」当元数据值吃掉（后面跟着「、」= 正文）。
    """
    return (s == '' or bool(_EXC_LABEL_RE.match(s)) or bool(_EXC_TIME_RE.match(s))
            or bool(_EXC_GLYPH_RE.match(s)) or bool(_EXC_BAR_RE.match(s)))


def _strip_excerpt_boiler(t):
    """吃掉摘要开头的「标签：短值」序列，判定保守 —— 宁可不剥也不吞正文。

    规则：① 值必须 ≤15 字、不含句子标点，且**后面必须跟标签/时间/串尾**才认（_exc_follow_ok）；
          ② 整个前缀必须含**日期**或**≥2 个标签**才提交 ——
             单纯一个「来源：」不动，避免把「来源：本公司拟投资…」这类正文句子当元数据吃掉。
    """
    pos = nlab = 0
    while pos < len(t) and pos <= 80:
        rest = t[pos:]
        m = _EXC_LABEL_RE.match(rest)
        if m:
            nlab += 1
            pos += m.end()
            rest2 = t[pos:]
            # ⚠️ 紧跟另一个标签（`点击数： 字号：`）→ 本标签是空值，别把下一个标签名当值吃掉
            if _EXC_LABEL_RE.match(rest2):
                continue
            # 顺序要紧：先试「T | T」字号条，再试短值 —— 否则值正则会先吃掉孤零零的 `T`
            gm = _EXC_GLYPH_RE.match(rest2)
            if gm:
                pos += gm.end()
                continue
            vm = _EXC_VAL_RE.match(rest2)
            if vm and _exc_follow_ok(rest2[vm.end():].lstrip()):
                pos += vm.end()
            continue
        for _re in (_EXC_TIME_RE, _EXC_WS_RE, _EXC_GLYPH_RE, _EXC_BAR_RE, _EXC_JUNK_RE):
            m = _re.match(t[pos:])
            if m:
                pos += m.end()
                break
        else:
            # 不带值的噪声标记（`微博 QQ空间`、`[小 中 大]`、`XX门户网站：URL`）也算一个标签
            m = _EXC_BARE_RE.match(t[pos:]) or _EXC_SITE_RE.match(t[pos:])
            if m:
                nlab += 1
                pos += m.end()
                continue
            break
    if nlab == 0 or (nlab < 2 and not _EXC_DATE_RE.search(t[:pos])):
        return t
    return t[pos:].lstrip(' |｜·')


def body_is_html(s):
    """True = 含真实 HTML 标签（按 HTML 渲染）；False = 纯文本（用 .ct，保留 \\n\\n 分段）"""
    return bool(_BODY_TAG_RE.search(s or ''))


# ─── 投资额抽取（2026-09-11）────────────────────────────────────────────
# 从正文里正则抽投资额，渲染时用（gov_raw 没有投资列，只能从正文取）。
# 精度优先，宁可抽不到也不给错值：
#   * 单位**必须带「元」**（万元/亿元）—— 只用「万」会被「年产15000万条」「5万吨」骗到；
#     实测这个约束干掉了 `投资建设"年产15000万条复合编织袋"` 这类假阳性。
#   * 先把 HTML 标签替换成空格再匹配 —— 标签会把数字切开
#     （`本项目总投资</span><span>158</span>…万元`），不剥就漏。
#   * 多处命中时优先「总投资 / 投资总额 / 总投资额 / 投资规模」类，
#     避免被前面出现的「环保投资」小项带偏（实测：`总投资 158 万元，其中环保投资…`
#     正确取到 158）。
#   * 实测命中率约 8.5%（400 条最新正文样本）；未命中的里面 43/45 条正文根本
#     不含投资词（如「当月支出资金440.06万元」是社会救助支出），属正确跳过。
_INVEST_RE = re.compile(
    r'(?:项目)?(?:估算)?(?:总)?(?:投资规模|投资总额|总投资额|投资额|投资估算|建设投资|工程投资|投资)'
    r'[^0-9<>\n]{0,8}'
    r'([0-9][0-9,，]*(?:\.[0-9]+)?)\s*(亿元|万元)')
_INVEST_PREFER_RE = re.compile(r'总投资|投资总额|总投资额|投资规模')


def extract_investment(content):
    """正文 → '1010万元' / '87.2亿元'；抽不到返回 ''（调用方据此决定是否显示该项）"""
    t = re.sub(r'<[^>]{0,300}>', ' ', content or '')
    ms = list(_INVEST_RE.finditer(t))
    if not ms:
        return ''
    m = next((x for x in ms if _INVEST_PREFER_RE.search(x.group(0))), ms[0])
    return m.group(1).replace(',', '').replace('，', '') + m.group(2)


DETAIL_CSS = ('.dh{background:#fff;border-radius:8px;padding:24px;margin-bottom:16px;box-shadow:0 1px 3px rgba(0,0,0,.08)}.dh h1{font-size:28px;margin:0;color:#1a1a2e}.dht{display:flex;align-items:flex-start;justify-content:space-between;gap:16px;padding-bottom:12px;margin-bottom:14px;border-bottom:1px solid #eef1f5}.dht h1{flex:1 1 auto;min-width:0}.dht-v{flex:0 0 auto;font-size:12.5px;color:#9096a0;line-height:1.7;white-space:nowrap}.dg{display:grid;grid-template-columns:1fr 1fr;gap:8px 32px;font-size:14px}.lb{color:#888;min-width:90px;flex-shrink:0}.ds{background:#fff;border-radius:8px;padding:20px 24px;margin-bottom:16px;box-shadow:0 1px 3px rgba(0,0,0,.08)}.ds h3{font-size:16px;margin-bottom:10px;padding-bottom:6px;border-bottom:2px solid #4361ee}.d-meta{display:flex;flex-wrap:wrap;align-items:baseline;gap:4px 18px;margin-top:12px;padding-top:12px;border-top:1px solid #eef1f5;font-size:13.5px;color:#5f6368}.d-meta .mi{display:inline-flex;align-items:baseline;gap:2px;max-width:100%;overflow-wrap:break-word}.d-meta .lb{min-width:0;color:#888}.d-meta .lnk{color:#1a73e8;text-decoration:none;font-weight:600}.d-meta .lnk:hover{text-decoration:underline}.d-meta .inv{color:#1a1a2e;font-weight:600}.d-meta .vis{color:#9096a0;font-size:12.5px}.d-visits{margin-top:10px;padding-top:10px;border-top:1px solid #eef1f5;color:#9096a0;font-size:12.5px}.ct{font-size:14px;line-height:1.8;color:#444;max-width:100%;overflow-wrap:break-word;word-break:break-word;overflow-x:auto;white-space:pre-line}.ct table{border-collapse:collapse;width:100%;margin:14px 0;font-size:14px}.ct td,.ct th{border:1px solid #ddd;padding:8px 12px;text-align:left;vertical-align:top}.ct tr:nth-child(even){background:#f9f9f9}.ct p{margin:12px 0;line-height:1.8}.ct-html{font-size:14px;line-height:1.8;color:#444;max-width:100%;overflow-wrap:break-word;word-break:break-word;overflow-x:auto}.ct-html p{margin:6px 0;line-height:1.8}.ct-html table{border-collapse:collapse;width:100%;margin:14px 0;font-size:14px}.ct-html td,.ct-html th{border:1px solid #ddd;padding:8px 12px;text-align:left;vertical-align:top}.ct-html tr:nth-child(even){background:#f9f9f9}.ct-html a{color:#1a73e8;text-decoration:none}.ct-html img{max-width:100%;height:auto;display:block;margin:8px 0;border-radius:8px}.ct a{color:#1a73e8;text-decoration:none}.ct img{max-width:100%;height:auto;display:block;margin:8px 0;border-radius:8px}.ds{overflow:hidden}.cc{border:1px solid #e8e8e8;border-radius:8px;padding:16px;margin-bottom:12px;background:#fafafa}.cc .cm{font-size:15px;font-weight:600;margin-bottom:4px}.cc .fd{display:grid;grid-template-columns:auto 1fr;gap:4px 12px;font-size:14px;margin-top:8px}.cl{color:#888}')

PAGE_SHELL_CSS = ('body{font-family:-apple-system,"Microsoft YaHei",sans-serif;background:#f1f3f5;color:#333;margin:0}.wrap{max-width:1000px;margin:0 auto;padding:20px 20px 56px}.bk-bar{background:#fff;border:1px solid #e9ecef;border-radius:12px;padding:10px 16px;margin-bottom:16px;box-shadow:0 1px 3px rgba(0,0,0,.05);display:flex;align-items:center;justify-content:space-between;gap:12px;flex-wrap:wrap}a{color:#4361ee}.bk{display:inline-block;font-size:14px;text-decoration:none}')


def render_project_detail_body(db_type, p, pv_total=0, username=""):
    """渲染项目正文卡片（详情页与列表页右侧抽屉共用，2026-09-10 抽出）。

    不含 <html>/<style>/页壳 —— 详情页 ?frag=1 或 /api/item 拿到的就是本片段。
    """
    h = ''
    # Detail card header

    proj_name = esc(p.get('project_name', p.get('title', p.get('owner_company', ''))))

    # 2026-09-11：标题单独放进 .dht 容器（原来是 <h1> 直接贴 .dg 网格，间距只有
    # 10px，视觉上挤在一起），下方一条细线把标题与下面的信息分开。
    # 「N 次访问」**移出标题行**（用户反馈挤在标题旁）→ crawler 作为元信息行最后一项，
    # 其它路由放在字段网格下方一行小字（.d-visits）。
    h += '<div class="dh"><div class="dht"><h1>' + proj_name + '</h1></div>'
    h += '<div class="dg">'



    # Render fields

    if db_type == 'znlh':

        for lbl,key in [('项目编号','project_id'),('当前阶段','phase'),('投资','budget'),('行业','industry'),('区域','region'),('省份','province'),('城市','city'),('工程类型','nature'),('占地面积','area'),('采购情况','purchase_status'),('竣工日期','completion_date'),('权重','weight'),('业主类型','owner_type'),('发布日期','publish_date'),('资金情况','funding_status')]:

            v = p.get(key,'')

            if v and v not in ('一一','——'):
                if key == 'industry':
                    v = ind_cn(v)
                h += '<div><span class="lb">'+lbl+'：</span><span>'+esc(v)+'</span></div>'

    elif db_type != 'crawler':

        # crawler 不用这张网格：gov_raw 表里只有 publish_date / industry 两列会命中
        # 下面这个列表（其余列在 gov_raw 里根本不存在），而这两项已经搬进 .d-meta
        # 元信息行，所以 crawler 直接跳过整张网格（残留的空 .dg 盒子渲染高度为 0）。
        for lbl,key in [('项目编号','project_id'),('版本类型','version_type'),('发布时间','publish_date'),('项目阶段','phase'),('建设周期','construction_period'),('总投资','total_investment'),('工程类型','project_type'),('甲方性质','owner_nature'),('行业','industry'),('规模','scale'),('省份','province'),('城市','city'),('详细地址','detail_address'),('建筑面积','building_area'),('占地面积','land_area'),('装修','decoration'),('电梯','elevator'),('空调','air_conditioning')]:

            v = p.get(key,'')

            if v and v not in ('一一','——'):
                if key == 'industry':
                    v = ind_cn(v)
                h += '<div><span class="lb">'+lbl+'：</span><span>'+esc(v)+'</span></div>'



    # 只关 .dg，.dh 留到下面一起关。原文这里多关了一层 → .dh 提前闭合，
    # 来源/原文链接被甩到白卡外面，末尾还多出 2 个游离 </div>。
    h += '</div>'



    # 元信息行（2026-09-11，crawler 专用）：按用户口径的顺序压成一行 ——
    #   行业 → 来源 → 发布时间 → 原文链接 → 投资额 → N 次访问
    #   ① 不再单独渲染「发布日期」（与「发布时间」同一个值，原先靠 .dg 网格判重，
    #      现在 crawler 不渲染网格了，直接用发布时间即可）；
    #   ② 原文链接把真实 URL 内嵌到「原文链接」四个字上，不铺裸 URL；
    #   ③ 投资额 = extract_investment() 从正文正则抽取，抽不到就不显示该项；
    #   ④ 「N 次访问」从标题行移到这里，作为行尾最弱的一项。

    if db_type == 'crawler':

        _bits = []
        # 「行业：other」= 未分类，属于无信息量字段 → 不显示（与列表 chips 的 other 口径一致）
        if p.get('industry') and (p['industry'] or '').strip().lower() != 'other':
            _bits.append('<span class="mi"><span class="lb">行业：</span><span>'
                         + esc(ind_cn(p['industry'])) + '</span></span>')
        if p.get('domain'):
            _bits.append('<span class="mi"><span class="lb">来源：</span><span>'
                         + esc(p['domain']) + '</span></span>')
        if p.get('publish_date'):
            _bits.append('<span class="mi"><span class="lb">发布时间：</span><span>'
                         + esc(p['publish_date']) + '</span></span>')
        if p.get('url'):
            _bits.append('<a class="mi lnk" href="' + esc(p['url'])
                         + '" target="_blank" rel="noopener">原文链接</a>')
        _inv = extract_investment(p.get('content'))
        if _inv:
            _bits.append('<span class="mi"><span class="lb">投资额：</span>'
                         '<span class="inv">' + esc(_inv) + '</span></span>')
        if pv_total > 0:
            _bits.append('<span class="mi vis">' + str(pv_total) + ' 次访问</span>')
        if _bits:
            h += '<div class="d-meta">' + ''.join(_bits) + '</div>'

    elif pv_total > 0:

        h += '<div class="d-visits">' + str(pv_total) + ' 次访问</div>'

    h += '</div>'



    # Enterprise info (znlh)

    if db_type == 'znlh':

        for lbl,key in [('企业名称','owner_company'),('注册地址','registered_address'),('注册资金','registered_capital'),('企业负责人','company_contact'),('主营业务','business_scope')]:

            v = p.get(key,'')

            if v: h += '<div class="ds"><h3>'+lbl+'</h3><div class="ct">'+esc(v)+'</div></div>'

        for lbl,key in [('建设背景','background'),('建设内容','construction_content'),('工艺路线','process_route'),('建设地点','construction_location'),('社会效益','social_benefits')]:

            v = p.get(key,'')

            if v: h += '<div class="ds"><h3>'+lbl+'</h3><div class="ct">'+esc(v)+'</div></div>'

        for lbl,key in [('工艺设备','equipment_list'),('配套设施设备','supporting_equipment'),('客户所需设备','client_needed_equipment')]:

            v = p.get(key,'')

            if v: h += '<div class="ds"><h3>'+lbl+'</h3><div class="ct">'+esc(v)+'</div></div>'

        if p.get('project_progress'): h += '<div class="ds"><h3>项目进展</h3><div class="ct">'+esc(p['project_progress'])+'</div></div>'

        if p.get('detail'): h += '<div class="ds"><h3>特殊说明</h3><div class="ct">'+esc(p['detail'])+'</div></div>'

    else:

        if p.get('construction_content'): h += '<div class="ds"><h3>建设内容</h3><div class="ct">'+esc(p['construction_content'])+'</div></div>'

        if p.get('equipment_list'): h += '<div class="ds"><h3>设备清单</h3><div class="ct">'+esc(p['equipment_list'])+'</div></div>'

        if p.get('schedule_overview'): h += '<div class="ds"><h3>工期概述</h3><div class="ct">'+esc(p['schedule_overview'])+'</div></div>'



    # Crawler content - HTML to Markdown

    if db_type == 'crawler' and p.get('content'):
        _body = strip_body_noise(p['content'])    # ⚠️ 剥 <style>/<script>：原样注入会污染全页 CSS
        if _body.strip():
            h += ('<div class="ds"><h3>正文内容</h3><div class="'
                  + ('ct-html' if body_is_html(_body) else 'ct')
                  + '">' + _body + '</div></div>')

    # EIA content

    if db_type == 'eia':

        if p.get('pub_date'):

            h += '<div class="ds"><h3>发布日期</h3><div class="ct">' + esc(p['pub_date']) + '</div></div>'

        if p.get('site_name'):

            h += '<div class="ds"><h3>来源站点</h3><div class="ct">' + esc(p['site_name']) + '</div></div>'

        if p.get('url'):

            h += '<div class="ds"><h3>原文链接</h3><div class="ct"><a href="' + esc(p['url']) + '" target="_blank">' + esc(p['url']) + '</a></div></div>'

        if p.get('content'):

            _body = strip_body_noise(p['content'])
            if _body.strip():
                h += ('<div class="ds"><h3>正文内容</h3><div class="'
                      + ('ct-html' if body_is_html(_body) else 'ct')
                      + '">' + _body + '</div></div>')

    # Contacts

    contacts = p.get('contacts',[])

    if contacts:

        h += '<div class="ds"><h3>项目联系人</h3>'

        for c in contacts:

            h += '<div class="cc"><div class="cm">'+esc(c.get('company',''))+'</div><div class="fd">'

            if c.get('contact_name'): h += '<span class="cl">姓名：</span><span>'+esc(c['contact_name'])+'</span>'

            if c.get('department'): h += '<span class="cl">部门：</span><span>'+esc(c['department'])+'</span>'

            if c.get('position'): h += '<span class="cl">职务：</span><span>'+esc(c['position'])+'</span>'

            if c.get('phone'): h += '<span class="cl">手机：</span><span>'+esc(c['phone'])+'</span>'

            if c.get('address'): h += '<span class="cl">地址：</span><span>'+esc(c['address'])+'</span>'

            if c.get('remarks') and c['remarks'] not in ('一一','——',''): h += '<span class="cl">备注：</span><span>'+esc(c['remarks'])+'</span>'

            h += '</div></div>'

        h += '</div>'
    return h



import base64

# ══════════════════════════════════════════════════════════════════════════
# 列表页 v2：左列表 + 右详情抽屉 + 无限滚动 + 游标分页 + J/K + ⌘K + 预取
# 2026-09-10 落地。数据面 = keyset 游标（(publish_date, id) 双键，走 idx_gov_raw_date）
#   实测：日期+keyset 0.001s / bigram×n+keyset 0.030s / FTS-EXISTS+keyset 0.141s
#   ⚠️ 禁用 JOIN 写法 `JOIN gov_search ... ORDER BY r.publish_date DESC`（全排序 18.4s）；
#      JOIN 必须写成相关子查询 EXISTS，让日期索引驱动扫描。
# ══════════════════════════════════════════════════════════════════════════

LIST_DAYS = {'1d': 1, '3d': 3, '7d': 7, '30d': 30, '90d': 90, '1y': 365}


def _enc_cursor(pub, rid):
    try:
        return base64.urlsafe_b64encode(('%s|%s' % (pub or '', rid)).encode('utf-8')).decode().rstrip('=')
    except Exception:
        return ''


def _dec_cursor(cur):
    if not cur:
        return None, None
    try:
        s = base64.urlsafe_b64decode((cur + '=' * (-len(cur) % 4)).encode()).decode('utf-8', 'ignore')
        pub, rid = s.rsplit('|', 1)
        return pub, int(rid)
    except Exception:
        return None, None


# 实体名特征后缀（完整公司/机构全称）：命中 gov_entity 精确索引。
# 正文里的公司名 gov_search 看不见——gov_search 是 trigram 分词且只索引
# title/site_name/summary 三列。
_ENTITY_SUFFIXES = ('有限公司', '有限责任公司', '股份有限公司', '集团', '公司',
                    '研究院', '设计院', '大学', '医院', '银行', '事业部', '总局')
# bigram 全 gram 交集的上限（超过则回落 FTS；防用户粘贴超长句产生上百个 EXISTS）
_LIST_GRAM_CAP = 16
# FTS 命中规模探测阈值：> 此值判为"稠密"，改用相关 EXISTS（见 _fts_pred）
_FTS_DENSE_PROBE = 4000


def _fts_pred(term, cur):
    """按命中规模选 FTS 谓词形态 → (SQL 片段, 参数)。

    2026-09-10 实测：两种形态**互补**，不存在通吃的写法——
      · 稠密（>4000 命中）：相关 EXISTS `EXISTS(SELECT 1 FROM gov_search g WHERE g.rowid=r.id
        AND gov_search MATCH ?)` 沿日期索引扫、**凑满 31 行即停** → \"建设项目环境影响评价\"
        (11.1 万命中) 0.286s；此形态在稀疏词上会退化成全表扫（同形查询 271s / 35s / 27s）。
      · 稀疏（≤4000 命中）：`r.id IN (SELECT rowid FROM gov_search WHERE gov_search MATCH ?)`
        物化小集合 → \"环境影响报告书全文公示\"(70) 0.013s / 全称(793) 0.138s；
        此形态在稠密词上要物化 11 万 rowid → 12.4s。
    探测用 `LIMIT _FTS_DENSE_PROBE+1`：稠密词提前停止（便宜），稀疏词遍历小 doclist（也便宜）。
    """
    sq = sanitize_fts_query(term)
    # 无游标（无法探测）时的兜底选 IN 物化：最坏 12.4s（稠密词），
    # 而相关 EXISTS 兜底最坏 271s（稀疏词）——有游标的生产路径一定会探测，不受影响。
    frag = 'r.id IN (SELECT rowid FROM gov_search WHERE gov_search MATCH ?)'
    if cur is not None:
        try:
            n = len(cur.execute('SELECT rowid FROM gov_search WHERE gov_search MATCH ? LIMIT ?',
                                (sq, _FTS_DENSE_PROBE + 1)).fetchall())
            if n > _FTS_DENSE_PROBE:
                frag = 'EXISTS (SELECT 1 FROM gov_search g WHERE g.rowid=r.id AND gov_search MATCH ?)'
        except Exception:
            pass
    return frag, sq


def _list_term_preds(term, cur=None):
    """单个关键词 → (SQL 片段列表, 参数列表)；片段之间 AND。

    路径（2026-09-10 逐个实测选定，改之前先跑 scripts/keyset_bench.py）：
      1) **完整实体全称**（len≥6 且含 有限公司/集团/研究院…）→ gov_entity 精确索引
         `r.id IN (SELECT doc_id FROM gov_entity WHERE etype='company' AND value=?)` ∪ FTS。
         精确语义（用户原则：实体精确索引 > bigram），实测 0.09~0.20s
      2) 其他中文词（2-gram 数 ≤16）→ gov_bigram **全 gram 交集**：高频长词 0.001~0.021s、
         稀疏长词 0.43~0.67s、0 命中 0.36s（旧写法同词 271s）
      3) 长数字串（≥7 位，疑似电话）→ gov_entity(etype='phone') OR FTS
      4) 其他（英文/超长词）→ FTS，稠密/稀疏由 _fts_pred 自动选形态
      5) 无中文短词（<3 字）→ LIKE

    ⚠️ 三条实测铁律（违反任一条都是 10~270 倍级退化）：
      · gov_search **严禁**无条件用相关子查询 EXISTS(... MATCH ?)：稠密词快（0.29s）但
        稀疏词 271s；反之 IN 物化稀疏快（0.013s）而稠密 12.4s → 必须探测后二选一。
      · gov_entity **一律用 IN(...) 物化**，勿用相关 EXISTS（实体 0 命中时计划崩坏，
        6.5s → IN 0.006s）；且**必须带 etype**（不带 0.362s / 带 0.090s）。
      · **严禁** OR `r.title LIKE '%<词>%'`（实测 83s / 80s，全表逐行 LIKE）。
    """
    frags, params = [], []
    t = (term or '').strip()
    if not t:
        return frags, params
    grams = token_grams(t) if contains_chinese(t) else []
    ent_suffix = len(t) >= 6 and any(s in t for s in _ENTITY_SUFFIXES)
    if ent_suffix:
        # ① 完整实体全称 → gov_entity 精确索引 ∪ FTS（**精确语义**）。
        # 不用 bigram：2-gram 交集对全称会把非连续出现的字也配对（实测并集 2036，
        # 比 entity 1373 多出 663 条假阳性）。用户既有原则：实体精确索引 > bigram。
        f, sq = _fts_pred(t, cur)
        frags.append("(r.id IN (SELECT doc_id FROM gov_entity WHERE etype='company' AND value=?) OR %s)" % f)
        params.extend([t, sq])
    elif grams and len(grams) <= _LIST_GRAM_CAP:
        # ② 其他中文词 → bigram 全 gram 交集（实测通吃长短词：高频长词 0.001~0.021s、
        #    稀疏长词 0.43~0.67s、0 命中 0.36s；2-gram 交集是"包含"的超集但无漏召）
        b = ' AND '.join(['EXISTS (SELECT 1 FROM gov_bigram b WHERE b.rowid=r.id AND b.gram=?)'] * len(grams))
        frags.append('(' + b + ')')
        params.extend(grams)
    elif len(re.sub(r'\D', '', t)) >= 7:
        f, sq = _fts_pred(t, cur)
        frags.append("(r.id IN (SELECT doc_id FROM gov_entity WHERE etype='phone' AND value=?) OR %s)" % f)
        params.extend([t, sq])
    elif len(t) >= 3:
        f, sq = _fts_pred(t, cur)
        frags.append(f)
        params.append(sq)
    else:
        frags.append('(r.title LIKE ? OR r.site_name LIKE ? OR r.page_url LIKE ?)')
        params.extend(['%' + t + '%'] * 3)
    return frags, params


def browse_items(cursor=None, limit=30, direction='next', daterange='all',
                 industry='', kw='', q='', with_total=False):
    """环评公示列表数据（keyset 游标分页，无限滚动专用）。

    返回 {items, next_cursor, prev_cursor, has_more, total}
    items 元素：{id, title, site, date, cat, ind, url, excerpt, has_table}
    """
    try:
        limit = max(1, min(int(limit or 30), 60))
    except Exception:
        limit = 30
    conn = get_db()
    conn.row_factory = sqlite3.Row
    c = conn.cursor()
    where, params = [], []

    days = LIST_DAYS.get(daterange)
    if days:
        # 用可走索引的字符串比较（date('now') 为字面量，索引范围扫描）+ GLOB 剔除脏格式
        where.append("r.publish_date >= date('now','localtime','-%d days') "
                     "AND r.publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*'" % days)
    _ind_frag, _ind_params = _ind_in_clause('r.industry', industry)
    if _ind_frag:
        where.append(_ind_frag)
        params.extend(_ind_params)

    # 用户关键词（AND）
    if (q or '').strip():
        for t in [x for x in re.split(r'[,\s]+', q.strip()) if x][:4]:
            f, p = _list_term_preds(t, c)
            where.extend(f)
            params.extend(p)
    # 管理员白名单关键词（OR）—— 与经典页共用 _whitelist_like_pred（2026-09-11 口径统一：
    # 原为 _list_term_preds 的 bigram 口径 = 标题+正文+摘要，与经典页不一致）
    _kw_frag, _kw_params = _whitelist_like_pred(
        [x for x in re.split(r'[,\s]+', (kw or '').strip()) if x][:8], 'r.')
    if _kw_frag:
        where.append(_kw_frag)
        params.extend(_kw_params)

    # keyset
    cd, ci = (None, None)
    if cursor:
        cd, ci = _dec_cursor(cursor)
    total = None
    total_truncated = False
    if cd is not None:
        if direction == 'prev':
            where.append('(r.publish_date, r.id) > (?, ?)')
            params.extend([cd, ci])
            order = 'r.publish_date ASC, r.id ASC'
        else:
            where.append('(r.publish_date, r.id) < (?, ?)')
            params.extend([cd, ci])
            order = 'r.publish_date DESC, r.id DESC'
    else:
        order = 'r.publish_date DESC, r.id DESC'
        if with_total:
            _kw_on = bool((kw or '').strip())
            try:
                if not (q or '').strip():
                    if not _kw_on:
                        # 日期/行业过滤都能走索引（idx_gov_raw_date / idx_gov_raw_industry）→ 精确 COUNT
                        total = c.execute('SELECT COUNT(*) FROM gov_raw r'
                                          + (' WHERE ' + ' AND '.join(where) if where else ''),
                                          params).fetchone()[0]
                    else:
                        # 白名单 LIKE 是全表扫（实测 **43s** / 244,839 条命中）→ 截断式计数：
                        # 沿日期索引流式、数到 5001 就停（**0.077s**），前端显示「5,001+」。
                        # 与 handle_db_list（经典页）同一手法、同一语义。
                        total = c.execute(
                            'SELECT COUNT(*) FROM (SELECT r.id FROM gov_raw r WHERE '
                            + ' AND '.join(where)
                            + ' ORDER BY r.publish_date DESC, r.id DESC LIMIT 5001)',
                            params).fetchone()[0]
                        total_truncated = total >= 5001
                # 带用户关键词 (q) 时不算总数（选择性不明，可能扫很久）→ 保持 "—"
            except Exception:
                total = None

    sql = ('SELECT r.id, r.title, r.site_name, r.publish_date, r.page_url, r.source_url, '
           'r.category, r.industry, r.has_table, SUBSTR(COALESCE(r.summary,""),1,300) AS _s, '
           # 2026-09-11 窗口 400 → 1200：政府页前 400 字常全是样板 div + h2 里的标题，
           # 真文本被截在窗口外 → 摘要空白。实测最新 200 条有实质摘要比例 88% → 92%
           # （3000 才 94%，8000 仍是 94% → 1200 是性价比拐点）。多取 800 字 × 25 行 ≈ 20KB/页，可忽略。
           'SUBSTR(COALESCE(r.content,""),1,1200) AS _c FROM gov_raw r'
           + (' WHERE ' + ' AND '.join(where) if where else '')
           + ' ORDER BY ' + order + ' LIMIT ?')
    try:
        c.execute(sql, params + [limit + 1])
        rows = [dict(x) for x in c.fetchall()]
    except Exception:
        rows = []
    conn.close()

    has_more = len(rows) > limit
    if has_more:
        rows = rows[:limit]
    if direction == 'prev':
        rows.reverse()   # 统一按 DESC 返回

    def _excerpt(r):
        raw = strip_body_noise(r.get('_c') or r.get('_s') or '')   # ⚠️ 必须先剥 <style>/<script>：只删标签会把块内 CSS 文本当正文
        txt = re.sub(r'<[^>]+>', ' ', raw)
        txt = re.sub(r'<[^>]*$', ' ', txt)        # ⚠️ SUBSTR 截断在标签中间 → 未闭合标签逃过上面正则，必须单独清
        txt = re.sub(r'&nbsp;|&amp;|&lt;|&gt;|&#\d+;|&[a-z]+;', ' ', txt)
        txt = re.sub(r'[<>"=]', ' ', txt)         # 残留尖括号/属性残片
        txt = re.sub(r'\s+', ' ', txt).strip()
        for _ in range(3):                        # 剥面包屑前缀（`导航菜单 `），可能连续多个
            _nxt = _EXC_LEAD_RE.sub('', txt)
            if _nxt == txt:
                break
            txt = _nxt
        txt = _strip_excerpt_boiler(txt)
        # 政府站常把标题原样抄在正文开头（标题已单独占一行）→ 去掉重复段，
        # 否则摘要 = 标题再来一遍，还把真文本挤出 180 字窗口。
        # ⚠️ 只认**开头整段**的重复。标题出现在句子中间时（实测「2026年9月9日，
        #    广汉市交通运输局组织召开…例会。」）去掉会把句子挖个洞 → 变成「2026年9月9日， 。」
        _t = (r.get('title') or '').strip()
        if len(_t) >= 8 and txt.startswith(_t):
            # 去掉标题后可能又露出页眉（`标题 发布时间：…`）→ 再剥一次
            txt = _strip_excerpt_boiler(txt[len(_t):].lstrip(' ·—-_|、'))
        return txt[:180]

    items = [{
        'id': str(r['id']),   # ⚠️ 必须字符串：gov_raw.id 是 19 位整数 > JS 安全整数(2^53)，数字形态会撞号
        'title': (r.get('title') or '').strip(),
        'site': (r.get('site_name') or '').strip(),
        'date': ((r.get('publish_date') or '')[:10]),
        # 2026-09-11 统一 item 形状：标签改成 [{v,c}]，前端 rowEl 只认这一套，
        # 从而 /zc /ccpc /contact 复用同一份 LIST_UI_JS 而不需要 per-route 分支。
        'tags': ([{'v': ind_cn((r.get('industry') or '').strip()), 'c': 'ind'}]
                 if (r.get('industry') or '').strip() not in ('', 'other') else [])
                + ([{'v': (r.get('category') or '').strip(), 'c': ''}]
                   if (r.get('category') or '').strip() else [])
                + ([{'v': '表格', 'c': 'tb'}] if (r.get('has_table') or 0) else []),
        'cat': (r.get('category') or '').strip(),
        'ind': (r.get('industry') or '').strip(),
        'url': (r.get('page_url') or r.get('source_url') or '').strip(),
        'has_table': r.get('has_table') or 0,
        'excerpt': _excerpt(r),
    } for r in rows]
    nxt = _enc_cursor(items[-1]['date'], items[-1]['id']) if items else ''
    prv = _enc_cursor(items[0]['date'], items[0]['id']) if items else ''
    return {'items': items, 'next_cursor': nxt if has_more else '', 'prev_cursor': prv,
            'has_more': has_more, 'total': total, 'total_truncated': total_truncated,
            'count': len(items)}


def list_quicksearch(q, limit=12):
    """⌘K 快速搜索（crawler）。返回精简字段，**不做精确计数**。

    2026-09-11 性能重写（实测 28~48s → 0.01~0.7s）：
    原实现直接调 search()，中文大词（「环境影响评价」命中 12.8 万行）会
      ① 先跑一次 JOIN gov_raw 的全量 COUNT（只为显示总数，12 万次回表），
      ② 再 ORDER BY r.publish_date DESC 对**全部命中**回表排序；
    机器上有爬虫并发写 search.db 时实测 28~48s → ⌘K 弹层卡死（48s 才出结果）。
    现改用列表页 v2 已验证的 _list_term_preds（entity/bigram 索引；FTS 按命中规模
    自动二选 IN 物化 / 相关 EXISTS）+ publish_date 索引流式扫描，**凑满 N 条即停**。

    语义差异（刻意）：total 恒为 0 —— 不再为弹层数 12.8 万条；前端只在 total 为真时
    渲染「共 N 条匹配，前 N 条」，因此本路由退化为不显示该计数提示（其余路由照旧）。
    另：不套管理页白名单 —— 与改动前一致（原实现也没套），⌘K 是导航入口不是列表页。
    """
    q = (q or '').strip()
    if len(q) < 2:
        return {'items': [], 'total': 0}
    try:
        limit = max(1, min(int(limit or 12), 30))
    except Exception:
        limit = 12
    out = []
    try:
        conn = sqlite3.connect(DB_PATH, timeout=10)
        conn.row_factory = sqlite3.Row
        c = conn.cursor()
        try:
            where, params = [], []
            for t in [x for x in re.split(r'[,\s]+', q) if x][:4]:
                f, p = _list_term_preds(t, c)
                where.extend(f)
                params.extend(p)
            def _fetch(where_list, plist):
                if not where_list:
                    return
                c.execute('SELECT r.id, r.title, r.site_name, r.publish_date FROM gov_raw r'
                          ' WHERE ' + ' AND '.join(where_list)
                          + ' ORDER BY r.publish_date DESC LIMIT ?', plist + [limit])
                for r in c.fetchall():
                    out.append({
                        # ⚠️ 必须字符串：gov_raw.id 是 19 位整数 > JS 安全整数(2^53)，数字形态会撞号
                        'id': str(r['id']),
                        'title': (r['title'] or '').strip()[:120],
                        'site': (r['site_name'] or '').strip()[:40],
                        'date': (r['publish_date'] or '')[:10],
                    })

            _fetch(where, params)
            if not out:
                # 兜底（只在快路径 0 命中时跑，不影响常态）：gram 索引打不中时退回 FTS 全文
                # （含正文）。_fts_pred 会先探测命中规模，再选 IN 物化 / 相关 EXISTS 形态，
                # 两种词都便宜（2026-09-11 实测 0.00~0.08s）。
                _fw, _fp = [], []
                for t in [x for x in re.split(r'[,\s]+', q) if x][:4]:
                    _f, _p = _fts_pred(t, c)
                    _fw.append('(' + _f + ')')
                    _fp.append(_p)
                _fetch(_fw, _fp)
        finally:
            conn.close()
    except Exception:
        out = []
    return {'items': out, 'total': 0}


LIST_UI_CSS = """<style>
*{box-sizing:border-box}
body{margin:0;background:#f1f3f5;color:#202124;font-family:-apple-system,BlinkMacSystemFont,"Microsoft YaHei","PingFang SC",sans-serif;overflow:hidden}
.lp-head{height:58px;display:flex;align-items:center;gap:14px;padding:0 18px;background:#fff;border-bottom:1px solid #e6e9ed;position:fixed;top:0;left:0;right:0;z-index:60}
.lp-logo{font-weight:700;color:#1a73e8;font-size:17px;text-decoration:none;letter-spacing:-.4px}
.lp-search{flex:1;max-width:520px;position:relative}
.lp-search input{width:100%;padding:9px 74px 9px 14px;border:1px solid #dfe3e8;border-radius:10px;font-size:13.5px;outline:none;background:#f8f9fb}
.lp-search input:focus{border-color:#1a73e8;background:#fff;box-shadow:0 0 0 3px rgba(26,115,232,.12)}
.lp-search .lp-kbd{position:absolute;right:9px;top:6px;font-size:11px;color:#8a9099;border:1px solid #e0e4e9;border-radius:6px;padding:2px 7px;background:#fff}
.lp-chips{display:flex;gap:5px;align-items:center;flex-wrap:nowrap}
.lp-chip{padding:5px 11px;border:1px solid #dfe3e8;border-radius:999px;background:#fff;font-size:12.5px;color:#3c4043;text-decoration:none;cursor:pointer;white-space:nowrap}
.lp-chip:hover{background:#eef4ff;border-color:#1a73e8;color:#1a73e8}
.lp-chip.act{background:#1a73e8;border-color:#1a73e8;color:#fff;font-weight:600}
.lp-ind{padding:6px 9px;border:1px solid #dfe3e8;border-radius:9px;font-size:12.5px;background:#fff;outline:none;max-width:150px}
/* ── 行业多选（按钮 + 弹层复选框）── */
.lp-indwrap{position:relative}
.lp-indbtn{width:100%;padding:7px 10px;border:1px solid #dfe3e8;border-radius:9px;font-size:12.5px;background:#fff;cursor:pointer;color:#3c4043;text-align:left;white-space:nowrap;overflow:hidden;text-overflow:ellipsis;font-family:inherit}
.lp-indbtn:hover{background:#eef4ff;border-color:#1a73e8;color:#1a73e8}
.lp-indbtn.on{border-color:#1a73e8;color:#1a73e8;font-weight:600;background:#eef4ff}
.lp-indpop{position:absolute;top:calc(100% + 6px);right:0;width:280px;max-height:58vh;overflow-y:auto;background:#fff;border:1px solid #e3e7ec;border-radius:12px;box-shadow:0 12px 34px rgba(20,30,50,.18);padding:8px;z-index:150}
.lp-indpop-hd{display:flex;justify-content:space-between;align-items:center;font-size:12px;color:#7a8089;padding:4px 6px 8px;border-bottom:1px solid #f0f2f5;margin-bottom:6px}
.lp-indpop-hd a{color:#1a73e8;text-decoration:none}
.lp-indit{display:flex;align-items:center;gap:8px;padding:7px 8px;border-radius:8px;font-size:13px;cursor:pointer;-webkit-user-select:none;user-select:none}
.lp-indit:hover{background:#f2f7ff}
.lp-indpop label.lp-indit input[type=checkbox]{-webkit-appearance:none;appearance:none;width:16px;height:16px;min-width:16px;margin:0;padding:0;flex:0 0 auto;border:1.5px solid #c3c9d2;border-radius:4px;background:#fff;box-shadow:none;cursor:pointer;position:relative;transition:border-color .12s,background .12s}
.lp-indpop label.lp-indit input[type=checkbox]:hover{border-color:#1a73e8}
.lp-indpop label.lp-indit input[type=checkbox]:checked{background:#1a73e8;border-color:#1a73e8}
.lp-indpop label.lp-indit input[type=checkbox]:checked:after{content:"";position:absolute;left:4.5px;top:1.5px;width:4px;height:8px;border:solid #fff;border-width:0 2px 2px 0;transform:rotate(45deg)}
.lp-indpop label.lp-indit input[type=checkbox]:focus{box-shadow:0 0 0 3px rgba(26,115,232,.18)}
.lp-right{margin-left:auto}
.lp-main{position:fixed;top:58px;left:0;right:0;bottom:0;display:flex}
.lp-list{flex:1;min-width:0;overflow-y:auto;padding:14px 16px 80px;scroll-behavior:auto}
.lp-list-inner{max-width:880px;margin:0 auto}
.lp-meta-bar{max-width:880px;margin:0 auto 10px;font-size:12.5px;color:#7a8089;display:flex;align-items:center;gap:10px}
.lp-item{background:#fff;border:1px solid #e9ecf1;border-left:3px solid transparent;border-radius:11px;padding:11px 14px;margin-bottom:8px;cursor:pointer;transition:box-shadow .12s,border-color .12s,background .12s}
.lp-item:hover{border-color:#cfd8e3;box-shadow:0 2px 10px rgba(20,30,50,.07)}
.lp-item.sel{border-left-color:#1a73e8;background:#f7faff;border-color:#c9dcfb;box-shadow:0 2px 10px rgba(26,115,232,.10)}
.lp-item .t{font-size:14.5px;font-weight:600;color:#1a1a2e;line-height:1.42;display:-webkit-box;-webkit-line-clamp:2;-webkit-box-orient:vertical;overflow:hidden}
.lp-item .m{font-size:12px;color:#7a8089;margin-top:5px;display:flex;gap:10px;flex-wrap:wrap;align-items:center}
.lp-item .e{font-size:12.5px;color:#5f6368;margin-top:5px;line-height:1.5;display:-webkit-box;-webkit-line-clamp:2;-webkit-box-orient:vertical;overflow:hidden}
.lp-tag{padding:1px 7px;border-radius:6px;background:#eef2f7;color:#4a5568;font-size:11px}
.lp-tag.ind{background:#eaf7ee;color:#1e7d45}
.lp-tag.tb{background:#fff3e0;color:#b26a00}
.lp-skel{height:74px;border-radius:11px;background:linear-gradient(90deg,#f2f4f7 25%,#e9edf2 37%,#f2f4f7 63%);background-size:400% 100%;animation:lpsk 1.2s ease infinite;margin-bottom:8px}
@keyframes lpsk{0%{background-position:100% 50%}100%{background-position:0 50%}}
.lp-drawer{width:46%;max-width:820px;min-width:420px;background:#f1f3f5;border-left:1px solid #e3e7ec;overflow-y:auto;padding:16px 18px 80px;display:none}
.lp-drawer.open{display:block}
.lp-drawer .d-close{position:sticky;top:0;float:right;z-index:5;border:1px solid #dfe3e8;background:#fff;border-radius:8px;padding:4px 10px;font-size:12px;cursor:pointer;color:#5f6368}
/* 2026-09-11: 抽屉顶部不再渲染标题（renderDrawer 已去掉 .d-title）——
   正文卡片第一行就是同一个标题，两份叠着显示是重复。
   这里把正文标题从 28px 收到 20px（≈二级标题），与正文 14px 的落差不再突兀。
   仅作用于列表页抽屉，独立详情页仍用 DETAIL_CSS 的 28px。 */
.lp-drawer .dh h1{font-size:20px}
.lp-empty{text-align:center;color:#8a9099;font-size:13.5px;padding:60px 10px}
.lp-more{max-width:880px;margin:6px auto 0;text-align:center;color:#8a9099;font-size:12.5px;padding:12px}
.lp-hint{position:fixed;left:18px;bottom:12px;background:rgba(32,33,36,.86);color:#e8eaed;font-size:11.5px;padding:6px 12px;border-radius:9px;z-index:70;display:flex;gap:12px;align-items:center}
.lp-hint kbd{background:#3c4043;border-radius:4px;padding:1px 5px;font-family:inherit;font-size:11px}
.lp-bar{position:fixed;top:0;left:0;height:2px;width:0;background:#1a73e8;z-index:99;transition:width .25s}
.lp-cmk{position:fixed;inset:0;background:rgba(20,24,32,.42);z-index:200;display:none;align-items:flex-start;justify-content:center;padding-top:12vh}
.lp-cmk.open{display:flex}
.lp-cmk-box{width:620px;max-width:92vw;background:#fff;border-radius:14px;box-shadow:0 18px 60px rgba(0,0,0,.28);overflow:hidden}
.lp-cmk-box input{width:100%;border:none;border-bottom:1px solid #eceff3;padding:15px 18px;font-size:15px;outline:none}
.lp-cmk-res{max-height:52vh;overflow-y:auto}
.lp-cmk-it{padding:10px 18px;cursor:pointer;border-bottom:1px solid #f4f6f8;font-size:13.5px}
.lp-cmk-it.act,.lp-cmk-it:hover{background:#f2f7ff}
.lp-cmk-it .s{font-size:11.5px;color:#8a9099;margin-top:3px}
.lp-cmk-foot{padding:9px 18px;font-size:11.5px;color:#8a9099;border-top:1px solid #eceff3;display:flex;gap:14px}
.lp-cmk-foot a{color:#1a73e8;text-decoration:none;margin-left:auto}
/* ── 手机端（≤760px）：抽屉=整屏详情页 ──
   ⚠️ 不加这段时：.lp-list{flex:1;min-width:0} 会把列表压成 0 宽，
   而 .lp-drawer{min-width:420px} 在 375/390px 屏上比屏幕还宽 → 右侧被裁 + 横向溢出。
   手机端呈"点一条就整屏变详情"，所以这里按整屏详情来排（去掉最小宽度与左边框）。 */
@media(max-width:760px){
  .lp-drawer{width:100%;max-width:none;min-width:0;border-left:none;padding:14px 14px 70px}
  .lp-list{padding:10px 10px 70px}
  .lp-item .e{display:none}                 /* 小屏不放摘要，避免一屏一条都看不全 */
  .lp-item .t{font-size:14px;-webkit-line-clamp:3}
  .lp-hint{display:none}                    /* 键盘提示在触屏上是噪音 */
  .lp-search{max-width:none}
  .lp-cmk-box{width:100%;max-width:100vw;border-radius:0}
  .lp-cmk{background:rgba(20,24,32,.5)}
  .lp-indpop{width:min(90vw,300px)}          /* 行业弹层在窄屏不超出屏幕 */
}
</style>"""


LIST_UI_JS = """<script>
(function(){
'use strict';
var BOOT = window.__BOOT__ || {};
var $ = function(s,r){return (r||document).querySelector(s)};
var itemsBox = $('#lp-items'), listBox = $('#lp-list'), drawer = $('#lp-drawer'),
    sentinel = $('#lp-sentinel'), bar = $('#lp-bar');
var S = { items: [], ids: {}, cursor: '', hasMore: true, loading: false, sel: null,
          busy: 0, filters: {range: BOOT.range || 'all', industry: BOOT.industry || '', q: BOOT.q || ''},
          cache: {}, cmkIdx: -1, cmkItems: [] };

function ping(on){ S.busy += on ? 1 : -1; if (S.busy < 0) S.busy = 0; bar.style.width = S.busy ? '72%' : '0'; if (!S.busy) setTimeout(function(){ bar.style.width='0' }, 200); }
function api(path, params){
  var qs = Object.keys(params||{}).filter(function(k){ return params[k] !== '' && params[k] != null })
           .map(function(k){ return k + '=' + encodeURIComponent(params[k]) }).join('&');
  return fetch(path + (qs ? '?' + qs : ''), {credentials:'same-origin'}).then(function(r){ return r.json() });
}
function esc(s){ return (s==null?'':String(s)).replace(/[&<>"']/g, function(c){ return {'&':'&amp;','<':'&lt;','>':'&gt;','"':'&quot;',"'":'&#39;'}[c] }) }

function rowEl(it){
  var d = document.createElement('div');
  d.className = 'lp-item'; d.dataset.id = it.id;
  var tags = (it.tags || []).map(function(t){
    return '<span class="lp-tag ' + (t.c || '') + '">' + esc(t.v) + '</span>' }).join('');
  d.innerHTML = '<div class="t">' + (esc(it.title) || '(无标题)') + '</div>'
    + '<div class="m">'
      + (it.site ? '<span>' + esc(it.site) + '</span>' : '')
      + (it.date ? '<span>' + esc(it.date) + '</span>' : '')
      + tags + '</div>'
    + (it.excerpt ? '<div class="e">' + esc(it.excerpt) + '</div>' : '');
  return d;
}

function appendItems(list){
  var frag = document.createDocumentFragment();
  list.forEach(function(it){
    var key = String(it.id);
    if (S.ids[key]) return;
    S.ids[key] = 1; it.id = key; S.items.push(it);
    frag.appendChild(rowEl(it));
  });
  itemsBox.appendChild(frag);
  $('#lp-count').textContent = S.items.length;
}

function skeleton(n){
  for (var i=0;i<n;i++){ var d=document.createElement('div'); d.className='lp-skel'; d.dataset.sk=1; itemsBox.appendChild(d); }
}
function clearSkel(){ Array.prototype.forEach.call(itemsBox.querySelectorAll('[data-sk]'), function(e){ e.remove() }) }

function loadMore(){
  if (S.loading || !S.hasMore) return Promise.resolve();
  S.loading = true; ping(true);
  if (!S.items.length) skeleton(4);
  return api('/api/list', {cursor:S.cursor, dir:'next', limit:30, db:'__ROUTE__', range:S.filters.range,
                           industry:S.filters.industry, q:S.filters.q})
    .then(function(d){
      clearSkel();
      if (!d || !d.ok){ S.hasMore = false; return }
      var _before = S.items.length, _cur0 = S.cursor;
      appendItems(d.items || []);
      S.cursor = d.next_cursor || ''; S.hasMore = !!d.has_more;
      // 守卫：游标没前进且没有新增 → 停止，避免无限循环重复请求
      if (S.cursor === _cur0 && S.items.length === _before) S.hasMore = false;
      if (d.total) $('#lp-total').textContent = d.total.toLocaleString() + (d.total_truncated ? '+' : '');
      if (!S.items.length) itemsBox.innerHTML = '<div class="lp-empty">没有匹配的记录<br><span style="font-size:12px">换个关键词或放宽时间范围试试</span></div>';
      else if (!S.hasMore) $('#lp-more').textContent = '— 已到底部，共 ' + S.items.length + ' 条 —';
    })
    .catch(function(){ clearSkel(); S.hasMore = false })
    .then(function(){ S.loading = false; ping(false); });
}

function resetList(){
  S.items = []; S.ids = {}; S.cursor = ''; S.hasMore = true; S.sel = null;
  itemsBox.innerHTML = ''; $('#lp-more').textContent = ''; $('#lp-count').textContent = '0';
  $('#lp-total').textContent = '—';
  drawer.classList.remove('open'); drawer.innerHTML = '';
  return loadMore();
}

// id → Promise（进行中的请求）。⚠️ 必须与"已完成结果"分开存：
// 旧写法把**占位符 1** 写进 S.cache，引发两个连锁缺陷 ——
//   ① prefetchItem() 复用分支 `if (S.cache[id]) return Promise.resolve(S.cache[id])`
//      会把占位符 1 当结果返回；
//   ② select() 里 `if (!cached)` 判定为"已有缓存"→ **跳过渲染回调** → 抽屉永久停在「加载中…」。
// 手机端无 hover，全靠 select() 自己预取相邻项(idx±1/idx+2)，连点相邻条目必踩
// （2026-09-10 实测：同一 tick 点第 30→31 条，3.3s 后仍 138 字节的「加载中」不恢复）。
S.pending = {};
function prefetchItem(id){
  id = String(id);
  if (typeof S.cache[id] === 'string') return Promise.resolve(S.cache[id]); // 已完成（含空串=失败）
  if (S.pending[id]) return S.pending[id];                                  // 进行中 → 复用同一 promise
  S.pending[id] = api('/api/item', {db:'__ROUTE__', id:id}).then(function(d){
    var h = (d && d.ok) ? (d.html || '') : '';
    S.cache[id] = h;
    return h;
  }).catch(function(){ S.cache[id] = ''; return ''; })
    .then(function(h){ delete S.pending[id]; return h; });
  return S.pending[id];
}
// 加载失败的重试入口（清掉失败缓存后重新选中）
window.__lpRetry = function(id){
  id = String(id); delete S.cache[id]; delete S.pending[id];
  select(id, true, false);
};
// 重试走事件委托（⚠️ 别在 JS 字符串里拼 javascript:__lpRetry('…')——
// Python 非 raw 字符串会把 \' 吃成 '，输出变成相邻字符串字面量 → 整块 SyntaxError）
drawer.addEventListener('click', function(e){
  var a = e.target.closest && e.target.closest('.lp-retry'); if (!a) return;
  e.preventDefault(); window.__lpRetry(a.getAttribute('data-retry'));
});

function renderDrawer(id, html){
  // 2026-09-11: 去掉抽屉顶部标题 —— 正文卡片第一行（.dh 里的 <h1>）就是同一个标题，
  // 初始打开和 J/K 翻页都走这里，两处一起消失。标题字号由 .lp-drawer .dh h1 控制。
  drawer.innerHTML = '<button class="d-close" onclick="window.__lpClose()">✕ 关闭</button>'
    + (html || '<div class="lp-empty">加载中…</div>');
  drawer.classList.add('open');
  drawer.scrollTop = 0;
}

function select(id, open, hist){
  var prev = S.sel; S.sel = String(id);
  Array.prototype.forEach.call(itemsBox.querySelectorAll('.lp-item.sel'), function(e){ e.classList.remove('sel') });
  var el = itemsBox.querySelector('.lp-item[data-id="' + String(id) + '"]');
  if (el) el.classList.add('sel');
  if (open !== false){
    var cached = S.cache[id];
    if (typeof cached === 'string' && cached){
      renderDrawer(id, cached);
    } else {
      renderDrawer(id, '');   // 占位：显示「加载中…」
      // ⚠️ 必须**无条件**挂 .then——cached 为 undefined 或"请求进行中"都要能回调重绘。
      //    旧写法 `if (!cached) prefetchItem(...).then(...)` 在请求进行中时直接跳过，
      //    导致抽屉永久停在「加载中」（手机端连点相邻条目必踩，见 prefetchItem 注释）。
      prefetchItem(id).then(function(h){
        if (String(S.sel) !== String(id)) return;   // 期间已切走 → 不覆盖
        renderDrawer(id, h || ('<div class="lp-empty">详情加载失败 '
          + '<a href="#" class="lp-retry" data-retry="' + String(id) + '">点此重试</a></div>'));
      });
    }
    // 预取相邻
    var idx = -1; for (var i=0;i<S.items.length;i++) if (String(S.items[i].id) === String(id)) { idx = i; break }
    if (idx >= 0){ [idx-1, idx+1, idx+2].forEach(function(k){ if (S.items[k]) prefetchItem(S.items[k].id) }) }
  }
  if (hist !== false){
    var st = {lp:1, id:String(id), scroll:listBox.scrollTop,
      f:{range:S.filters.range, industry:S.filters.industry, q:S.filters.q}, n:S.items.length};
    if (hist === 'replace') history.replaceState(st, '', location.pathname + location.search);
    else history.pushState(st, '', location.pathname + location.search);
  }
}
window.__lpClose = function(){ if (S.sel){ S.sel = null;
  Array.prototype.forEach.call(itemsBox.querySelectorAll('.lp-item.sel'), function(e){ e.classList.remove('sel') });
  drawer.classList.remove('open'); } };

// ── 无限滚动（提前 2 屏预取下一页：rootMargin 1200px）──
if ('IntersectionObserver' in window){
  new IntersectionObserver(function(es){ es.forEach(function(e){ if (e.isIntersecting) loadMore() }) },
    {root: listBox, rootMargin: '1200px 0px 1200px 0px'}).observe(sentinel);
} else {
  listBox.addEventListener('scroll', function(){ if (listBox.scrollTop + listBox.clientHeight + 900 > listBox.scrollHeight) loadMore() });
}

// ── 悬停预取详情 ──
var hoverTimer = null, hoverId = null;
itemsBox.addEventListener('mouseover', function(e){
  var el = e.target.closest && e.target.closest('.lp-item'); if (!el) return;
  hoverId = el.dataset.id; clearTimeout(hoverTimer);
  hoverTimer = setTimeout(function(){ prefetchItem(hoverId) }, 110);
});
itemsBox.addEventListener('mouseout', function(){ clearTimeout(hoverTimer) });

// ── 触屏预取（手机端没有 hover，指针按下即开始取详情）──
// 手指按下(pointerdown) 到 click 之间有 ~50-150ms，请求提前起飞，
// 配合 prefetchItem 的 promise 复用 → 点击时抽屉基本能立刻出内容。
itemsBox.addEventListener('pointerdown', function(e){
  if (e.pointerType === 'mouse') return;          // 鼠标走上面的 hover 逻辑
  var el = e.target.closest && e.target.closest('.lp-item'); if (!el) return;
  var id = el.getAttribute('data-id'); if (id) prefetchItem(id);
}, {passive: true});

// ── 点击打开 ──
itemsBox.addEventListener('click', function(e){
  var el = e.target.closest && e.target.closest('.lp-item'); if (!el) return;
  select(el.dataset.id, true, 'push');
});

// ── 空闲预取前 3 条详情 ──
function idlePrefetch(){
  var f = function(){ (S.items.slice(0,3)).forEach(function(it,i){ setTimeout(function(){ prefetchItem(it.id) }, i*220) }) };
  if (window.requestIdleCallback) requestIdleCallback(f, {timeout: 2200}); else setTimeout(f, 900);
}

// ── 键盘导航 ──
function move(delta){
  if (!S.items.length) return;
  var idx = -1; for (var i=0;i<S.items.length;i++) if (String(S.items[i].id) === String(S.sel)) { idx = i; break }
  idx += delta;
  if (idx < 0) idx = 0; if (idx > S.items.length-1) idx = S.items.length-1;
  var it = S.items[idx];
  if (idx >= S.items.length - 6) loadMore();
  select(it.id, true, 'replace');
  var el = itemsBox.querySelector('.lp-item[data-id="' + it.id + '"]');
  if (el){
    var r = el.getBoundingClientRect(), lr = listBox.getBoundingClientRect();
    if (r.top < lr.top + 10) listBox.scrollTop += r.top - lr.top - 10;
    else if (r.bottom > lr.bottom - 10) listBox.scrollTop += r.bottom - lr.bottom + 10;
  }
}
document.addEventListener('keydown', function(e){
  var t = e.target || {};
  var typing = /INPUT|TEXTAREA|SELECT/.test(t.tagName || '');
  var cmkOpen = $('#lp-cmk').classList.contains('open');
  if ((e.metaKey || e.ctrlKey) && (e.key === 'k' || e.key === 'K')){ e.preventDefault(); openCmk(); return }
  if (e.key === 'Escape'){ if (cmkOpen){ closeCmk(); return } window.__lpClose(); return }
  if (cmkOpen){ return }
  if (typing){ if (e.key === 'Enter' && t.id === 'lp-q'){ e.preventDefault(); applyQ(t.value); } return }
  if (e.key === 'j' || e.key === 'ArrowDown'){ e.preventDefault(); move(1) }
  else if (e.key === 'k' || e.key === 'ArrowUp'){ e.preventDefault(); move(-1) }
  else if (e.key === 'Enter'){ if (S.sel) select(S.sel, true, true) }
  else if (e.key === '/'){ e.preventDefault(); $('#lp-q').focus() }
});

// ── ⌘K 快速搜索 ──
function openCmk(){ $('#lp-cmk').classList.add('open'); var i = $('#lp-cmk-in'); i.value = ''; i.focus(); $('#lp-cmk-res').innerHTML = ''; S.cmkItems = []; S.cmkIdx = -1 }
function closeCmk(){ $('#lp-cmk').classList.remove('open') }
var cmkTimer = null;
function renderCmk(list, total){
  S.cmkItems = list || []; S.cmkIdx = list && list.length ? 0 : -1;
  var h = (list||[]).map(function(it,i){
    return '<div class="lp-cmk-it' + (i===0?' act':'') + '" data-i="' + i + '" data-id="' + it.id + '">'
      + esc(it.title) + '<div class="s">' + esc(it.site) + ' · ' + esc(it.date) + '</div></div>' }).join('');
  if (!h) h = '<div class="lp-cmk-it" style="color:#8a9099;cursor:default">无匹配结果</div>';
  $('#lp-cmk-res').innerHTML = h;
  $('#lp-cmk-foot-hint').textContent = total ? ('共 ' + total + ' 条匹配，前 ' + (list||[]).length + ' 条') : '';
}
$('#lp-cmk-in').addEventListener('input', function(){
  var v = this.value; clearTimeout(cmkTimer);
  if (!v.trim()){ renderCmk([],0); return }
  cmkTimer = setTimeout(function(){
    api('/api/quicksearch', {db:'__ROUTE__', q:v}).then(function(d){ renderCmk(d.items, d.total) });
  }, 160);
});
$('#lp-cmk-res').addEventListener('click', function(e){
  var el = e.target.closest && e.target.closest('.lp-cmk-it'); if (!el || !el.dataset.id) return;
  var seed = S.cmkItems[parseInt(el.dataset.i, 10)];
  closeCmk(); jumpTo(el.dataset.id, seed);
});
$('#lp-cmk-in').addEventListener('keydown', function(e){
  if (e.key === 'ArrowDown' || e.key === 'ArrowUp'){
    e.preventDefault();
    S.cmkIdx += (e.key === 'ArrowDown' ? 1 : -1);
    if (S.cmkIdx < 0) S.cmkIdx = 0; if (S.cmkIdx > S.cmkItems.length-1) S.cmkIdx = S.cmkItems.length-1;
    var els = $('#lp-cmk-res').querySelectorAll('.lp-cmk-it');
    Array.prototype.forEach.call(els, function(el,i){ el.classList.toggle('act', i === S.cmkIdx) });
    if (els[S.cmkIdx]) els[S.cmkIdx].scrollIntoView({block:'nearest'});
  } else if (e.key === 'Enter'){
    e.preventDefault();
    var it = S.cmkItems[S.cmkIdx];
    if (it){ closeCmk(); jumpTo(it.id, it) }
  }
});
$('#lp-cmk').addEventListener('click', function(e){ if (e.target === this) closeCmk() });
$('#lp-cmk-all').addEventListener('click', function(e){
  e.preventDefault(); var v = $('#lp-cmk-in').value.trim();
  if (v) location.href = '/?q=' + encodeURIComponent(v);
});
function jumpTo(id, seed){
  // 若已在列表里直接选中；否则把该条置顶插入
  id = String(id);
  seed = seed || {};
  if (S.ids[id]) { select(id, true, true); var el = itemsBox.querySelector('.lp-item[data-id="'+id+'"]'); if (el) el.scrollIntoView({block:'center'}); return }
  api('/api/item', {db:'__ROUTE__', id:id}).then(function(d){
    if (!d || !d.ok) return;
    var it = {id:id, title:d.title||seed.title||'', site:d.site||seed.site||'', date:d.date||seed.date||'', excerpt:d.excerpt||''};
    it.id = String(id); S.ids[id] = 1; S.items.unshift(it);
    itemsBox.insertBefore(rowEl(it), itemsBox.firstChild);
    prefetchItem(id); select(id, true, true);
    listBox.scrollTop = 0;
  });
}

// ── 筛选器 ──
function applyRange(v){ S.filters.range = v;
  Array.prototype.forEach.call(document.querySelectorAll('#lp-range .lp-chip'), function(a){ a.classList.toggle('act', a.dataset.v === v) });
  pushFilters(); resetList();
}
function applyQ(v){ S.filters.q = (v||'').trim(); $('#lp-q').value = S.filters.q; pushFilters(); resetList(); }
function pushFilters(){ history.pushState({lp:1, f:{range:S.filters.range, industry:S.filters.industry, q:S.filters.q}, scroll:0, n:0},
  '', '__BASE__' + '?range=' + S.filters.range + (S.filters.industry ? '&industry=' + encodeURIComponent(S.filters.industry) : '') + (S.filters.q ? '&q=' + encodeURIComponent(S.filters.q) : '')) }
document.querySelectorAll('#lp-range .lp-chip').forEach(function(a){ a.addEventListener('click', function(){ applyRange(this.dataset.v) }) });
// ── 行业多选（按钮 + 复选框弹层；2026-09-11 由单选 select 改为多选）──
// indSync 是函数声明 → 提升到 IIFE 顶部，popstate 里也能直接调（定义在后不影响）。
function indChecked(){
  var box = $('#lp-indpop'); if (!box) return [];
  return Array.prototype.filter.call(box.querySelectorAll('input'), function(cb){ return cb.checked })
         .map(function(cb){ return cb.value });
}
function indLabel(){
  var btn = $('#lp-indbtn'); if (!btn) return;
  var ids = indChecked();
  if (!ids.length){ btn.textContent = '全部行业 ▾'; btn.classList.remove('on'); return }
  if (ids.length === 1){
    var cb = $('#lp-indpop input[value="' + ids[0] + '"]');
    btn.textContent = (cb ? cb.parentNode.textContent.trim() : ids[0]) + ' ▾';
  } else { btn.textContent = '已选 ' + ids.length + ' 个行业 ▾' }
  btn.classList.add('on');
}
function indSync(){
  var box = $('#lp-indpop'); if (!box) return;
  var set = {};
  (S.filters.industry || '').split(/[,\s]+/).filter(Boolean).forEach(function(v){ set[v] = 1 });
  Array.prototype.forEach.call(box.querySelectorAll('input'), function(cb){ cb.checked = !!set[cb.value] });
  indLabel();
}
if ($('#lp-indbtn')){
  $('#lp-indbtn').addEventListener('click', function(e){
    e.stopPropagation(); var p = $('#lp-indpop');
    p.style.display = (p.style.display === 'block') ? 'none' : 'block';
  });
  $('#lp-indpop').addEventListener('click', function(e){ e.stopPropagation() });
  document.addEventListener('click', function(){ var p = $('#lp-indpop'); if (p) p.style.display = 'none' });
  $('#lp-indpop').addEventListener('change', function(){
    S.filters.industry = indChecked().join(','); indLabel(); pushFilters(); resetList();
  });
  $('#lp-indclr').addEventListener('click', function(e){
    e.preventDefault();
    Array.prototype.forEach.call($('#lp-indpop').querySelectorAll('input'), function(cb){ cb.checked = false });
    S.filters.industry = ''; indLabel(); pushFilters(); resetList();
  });
  indSync();
}
$('#lp-q').addEventListener('keydown', function(e){ if (e.key === 'Enter'){ e.preventDefault(); applyQ(this.value) } });

// ── 返回恢复（滚动/筛选/选中）──
window.addEventListener('popstate', function(e){
  var st = e.state || {};
  var f = st.f || {range:'all', industry:'', q:''};
  var changed = (f.range !== S.filters.range) || (f.industry !== S.filters.industry) || (f.q !== S.filters.q);
  if (changed){
    S.filters = {range:f.range||'all', industry:f.industry||'', q:f.q||''};
    $('#lp-q').value = S.filters.q;
    indSync();
    Array.prototype.forEach.call(document.querySelectorAll('#lp-range .lp-chip'), function(a){ a.classList.toggle('act', a.dataset.v === S.filters.range) });
    resetList().then(function(){ return restorePages(st.n) }).then(function(){ restore(st) });
  } else { restore(st) }
});
function restorePages(n, round){
  round = round || 0;
  if (!n || round > 8 || S.items.length >= n || !S.hasMore) return Promise.resolve();
  return loadMore().then(function(){ return restorePages(n, round + 1) });
}
function restore(st){
  listBox.scrollTop = st.scroll || 0;
  if (st.id && S.ids[String(st.id)]) select(st.id, true, false); else window.__lpClose();
}

// ── 启动 ──
if (BOOT.items && BOOT.items.length){ appendItems(BOOT.items); S.cursor = BOOT.cursor || ''; S.hasMore = !!BOOT.has_more;
  if (BOOT.total) $('#lp-total').textContent = BOOT.total.toLocaleString() + (BOOT.total_truncated ? '+' : ''); }
else { loadMore(); }
window.__lpDebug = function(){ return {items:S.items.length, sel:S.sel, cursor:S.cursor, hasMore:S.hasMore,
  cached:Object.keys(S.cache).filter(function(k){ return typeof S.cache[k] === 'string' && S.cache[k] }).length,
  cachedIds:Object.keys(S.cache).slice(0,8), loading:S.loading, filters:S.filters,
  scroll:listBox.scrollTop, drawerOpen:drawer.classList.contains('open'), drawerLen:drawer.innerHTML.length} };
history.replaceState({lp:1, f:S.filters, scroll:0}, '');
setTimeout(idlePrefetch, 500);
setTimeout(function(){ if (S.items.length && S.hasMore) loadMore() }, 260);   // 首屏后立刻补一页，滚动无等待
})();
</script>"""


def render_list_shell(username='', daterange='all', industry='', q='', route='crawler'):
    """列表页 v2 外壳（左列表 + 右抽屉）。首屏数据内联，避免二次请求。

    2026-09-11 多路由化：route ∈ {crawler, zc, ccpc, contact}。
    LIST_UI_JS 里的 __ROUTE__ / __BASE__ 占位符在**这里 str.replace 替换**——
    不能用 f-string：JS 里大量 {} 会被当成格式占位符。
    """
    meta = LIST_ROUTE_META.get(route, LIST_ROUTE_META['crawler'])
    # 2026-09-11 修：首屏与翻页**口径一致** —— 都走管理页白名单。
    # 旧版首屏不传 kw（只有 /api/list 传），导致"第一页不筛、往下滚才筛"，
    # 用户会以为白名单规则没生效。现在带上 kw，总数由 browse_items 的
    # **截断式计数**给出（0.077s，显示「5,001+」），不再退回 "—"。
    boot = browse_route(route, limit=30, daterange=daterange, industry=industry, q=q,
                        kw=(admin_whitelist_kw() if route == 'crawler' else ''), with_total=True)
    boot['range'] = daterange
    boot['industry'] = industry
    boot['q'] = q
    boot['route'] = route
    chips = [('all', '全部'), ('1d', '今日'), ('3d', '3天'), ('7d', '7天'), ('30d', '30天'), ('1y', '一年')]
    chips_html = ''.join(
        '<a class="lp-chip%s" data-v="%s" href="%s?range=%s">%s</a>' % (
            ' act' if daterange == v else '', v, meta['base'], v, label)
        for v, label in chips)
    # 行业筛选：**多选**（2026-09-11 由单选 select 改为「按钮 + 复选框弹层」——
    # <select multiple> 在手机上是灾难）。选中态由 URL 的 industry 逗号列表回填。
    _ind_sel = [x for x in re.split(r'[,\s]+', (industry or '').strip()) if x]
    _ind_items = list(INDUSTRIES) + [{'id': 'other', 'cn': '其他'}]
    if not _ind_sel:
        _ind_lbl, _ind_cls = '全部行业 ▾', ''
    elif len(_ind_sel) == 1:
        _ind_lbl = next((i['cn'] for i in _ind_items if i['id'] == _ind_sel[0]), _ind_sel[0]) + ' ▾'
        _ind_cls = ' on'
    else:
        _ind_lbl, _ind_cls = '已选 %d 个行业 ▾' % len(_ind_sel), ' on'
    ind_html = ('<button type="button" class="lp-indbtn' + _ind_cls + '" id="lp-indbtn">'
                + _ind_lbl + '</button>'
                '<div class="lp-indpop" id="lp-indpop" style="display:none">'
                '<div class="lp-indpop-hd"><b>行业（可多选）</b>'
                '<a href="#" id="lp-indclr">清空</a></div>'
                + ''.join('<label class="lp-indit"><input type="checkbox" value="%s"%s>%s</label>'
                          % (i['id'], ' checked' if i['id'] in _ind_sel else '', i['cn'])
                          for i in _ind_items)
                + '</div>')
    # 不支持该筛选的路由：元素保留在 DOM（JS 里的绑定无空指针风险），只把外层容器隐藏。
    _chips_v = '' if meta['chips'] else ' style="display:none"'
    # 2026-09-11 修复：原 'max-width:210px;flex:0 0 auto' 下按钮被挤成内容宽（实测 81px，
    # '已选 N 个行业 ▾' 直接进省略号）→ 加 min-width，使其是个正常「字段」。
    _ind_v = ' style="min-width:140px;max-width:210px;flex:0 0 auto"' if meta['ind'] else ' style="display:none"'
    return ('<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8">'
            '<meta name="viewport" content="width=device-width, initial-scale=1">'
            '<title>' + meta['title'] + ' · BJIIR项目库</title>'
            + LIST_UI_CSS + '<style>' + DETAIL_CSS + '</style>'
            '</head><body>'
            '<div class="lp-bar" id="lp-bar"></div>'
            '<header class="lp-head">'
            '<a class="lp-logo" href="/">BJIIR</a>'
            '<div class="lp-search"><input id="lp-q" placeholder="' + meta['ph'] + '" value="'
            + esc(q) + '"><span class="lp-kbd">⌘K</span></div>'
            '<div class="lp-chips" id="lp-range"' + _chips_v + '>' + chips_html + '</div>'
            '<div class="lp-search lp-indwrap"' + _ind_v + '>' + ind_html + '</div>'
            '<div class="lp-right">' + nav_right_html(username) + '</div>'
            '</header>'
            '<div class="lp-main">'
            '<section class="lp-list" id="lp-list"><div class="lp-list-inner">'
            '<div class="lp-meta-bar">已加载 <b id="lp-count">0</b> 条 · 共 <b id="lp-total">—</b> 条 '
            '<span style="color:#aab0b8">（' + meta['sort'] + ' · 无限滚动）</span></div>'
            '<div id="lp-items"></div>'
            '<div class="lp-more" id="lp-more"></div>'
            '<div id="lp-sentinel" style="height:1px"></div>'
            '</div></section>'
            '<aside class="lp-drawer" id="lp-drawer"></aside>'
            '</div>'
            '<div class="lp-hint"><span><kbd>J</kbd> <kbd>K</kbd> 上下移动</span>'
            '<span><kbd>Enter</kbd> 打开</span><span><kbd>⌘K</kbd> 搜索</span>'
            '<span><kbd>/</kbd> 筛选</span><span><kbd>Esc</kbd> 关闭</span></div>'
            '<div class="lp-cmk" id="lp-cmk"><div class="lp-cmk-box">'
            '<input id="lp-cmk-in" placeholder="快速搜索：标题 / 公司全称 / 关键词…（Esc 关闭）" autocomplete="off">'
            '<div class="lp-cmk-res" id="lp-cmk-res"></div>'
            '<div class="lp-cmk-foot"><span>↑↓ 选择</span><span>Enter 打开</span>'
            '<span id="lp-cmk-foot-hint"></span>'
            '<a href="/?q=" id="lp-cmk-all" onclick="return false">查看全部结果 →</a></div>'
            '</div></div>'
            '<script>window.__BOOT__ = ' + json.dumps(boot, ensure_ascii=False) + ';</script>'
            + LIST_UI_JS.replace('__ROUTE__', route).replace('__BASE__', meta['base']) +
            '</body></html>')


def admin_whitelist_kw():
    """读取管理页保存的白名单关键词（列表页与 /crawler 旧版保持一致）"""
    try:
        with open(ADMIN_EXCLUDE_PATH) as f:
            return f.read().strip()
    except Exception:
        return ""


# ══════════════════════════════════════════════════════════════════════════
#  列表页 v2 · 多路由支持（2026-09-11）
#  /crawler/ 的 app 式外壳（左列表 + 右抽屉 + 无限滚动 + 游标分页 + ⌘K + 预取）
#  由本块泛化到 /zc、/ccpc、/contact 三条路由。
#
#  设计要点：
#   · 所有路由的 browse_* 统一输出 item 形状
#       {id, title, site, date, tags:[{v,c}], excerpt, url}
#     前端 rowEl 只认这一套 → LIST_UI_JS 里没有 per-route 分支，只需
#     __ROUTE__ / __BASE__ 两个占位符在 render_list_shell 里 str.replace 替换。
#   · 游标用 _enc_curs/_dec_curs（**rid 保持字符串**）——不能复用 _enc_cursor /
#     _dec_cursor，那两个对 rid 做 int()，而 zc 主键 project_id 是 TEXT('ZC0001299697')。
#   · 三张表都是千级~万级行（1642 / 6117 / 19424），keyset 不依赖索引也是毫秒级。
#   · /contact 没有「发布时间」列 → date_col=None，退化成 id 单键 keyset，
#     同时列表页隐藏时间 chips（LIST_ROUTE_META['contact']['chips']=False）。
# ══════════════════════════════════════════════════════════════════════════

LIST_ROUTE_META = {
    'crawler': {'base': '/crawler/', 'title': '环评公示', 'chips': True, 'ind': True,
                'sort': '按发布时间倒序',
                'ph': '搜索项目标题/公司/关键词（≥3 字更准，回车筛选）'},
    'zc':      {'base': '/zc/', 'title': '中策大数据', 'chips': True, 'ind': False,
                'sort': '按发布时间倒序',
                'ph': '搜索项目名称 / 行业 / 地区（回车筛选）'},
    'ccpc':    {'base': '/ccpc', 'title': '中项网', 'chips': True, 'ind': False,
                'sort': '按发布时间倒序',
                'ph': '搜索项目名称 / 业主公司（回车筛选）'},
    'contact': {'base': '/contact', 'title': '联系人DB', 'chips': False, 'ind': False,
                'sort': '按录入倒序',
                'ph': '搜索公司 / 联系人 / 手机 / 邮箱（回车筛选）'},
}


def _whitelist_like_pred(kw_list, alias='r.'):
    """管理页白名单（"仅限给定关键词"）的**统一谓词**：标题/站点/URL 三列 LIKE，词间 OR。

    2026-09-11 口径统一。原来同一份白名单**两套语义**：
      · v2 列表 `browse_items`：`_list_term_preds` → 中文 2 字词走 `gov_bigram`
        （覆盖 **标题 + 正文 + 摘要**）
      · 经典页 `handle_db_list`：`title/site_name/page_url LIKE '%kw%'`
    实测同一条白名单 "项目 工程"：v2 语义首屏 30/30 合格、LIKE 口径只 23/30（7 条仅正文命中）
    → 同一份白名单两个页面出两套结果，**验证时极易误判"规则没生效"**。

    统一到本函数（三列 LIKE）的理由：
      1. 白名单是**公告级门禁** —— 该看公告自身的身份（标题 / 来源站 / URL），
         正文里顺带提到的关键词不该把一条公告放进来；
      2. 三列 LIKE 在 v2 与经典页的四条 SQL 路径上都能安全落地；换成 bigram 口径
         要给「无词浏览」路径加表别名（技能里明令的雷区），风险不成比例；
      3. 口径可被"看标题"直接验证 —— 这正是用户当初判定"没生效"的方式。
    代价：v2 列表不再包含"仅正文命中"的记录（本次实测约 7/30 条）。
    计数不受影响：白名单生效时仍走截断式计数（LIMIT 5001，0.077s → 显示 "5,001+"）。

    ⚠️ LIKE 不区分 ASCII 大小写，故调用方**不必**先 lowercase（经典页原来 lower() 过，
    结果等价）。返回 (None, []) 表示无白名单。
    """
    if not kw_list:
        return None, []
    frag = '(' + ' OR '.join(
        '(%stitle LIKE ? OR %ssite_name LIKE ? OR %spage_url LIKE ?)' % (alias, alias, alias)
        for _ in kw_list) + ')'
    return frag, ['%%%s%%' % str(k) for k in kw_list for _ in range(3)]


def _ind_in_clause(col, industry_csv):
    """industry 参数 → (SQL 片段, 参数列表)；**支持多选**（逗号/空格分隔，2026-09-11）。

    单选时代码是 `col = ?`，多选改 `col IN (?,?,...)`。
    去重保序；非法值原样进 IN（与旧行为等价：查不到即 0 条）。
    """
    ids = []
    for x in re.split(r'[,\s]+', (industry_csv or '').strip()):
        if x and x not in ids:
            ids.append(x)
    if not ids:
        return None, []
    return '%s IN (%s)' % (col, ','.join('?' * len(ids))), ids


def _enc_curs(pub, rid):
    """游标编码（跨非 crawler 路由共用；rid 保持字符串形态）。"""
    try:
        return base64.urlsafe_b64encode(('%s|%s' % (pub or '', rid)).encode('utf-8')).decode().rstrip('=')
    except Exception:
        return ''


def _dec_curs(cur):
    if not cur:
        return None, None
    try:
        s = base64.urlsafe_b64decode((cur + '=' * (-len(cur) % 4)).encode()).decode('utf-8', 'ignore')
        pub, rid = s.split('|', 1)
        return pub, rid
    except Exception:
        return None, None


def _clean_txt(v, n=180):
    """摘录清洗：去噪声块、去标签、去实体、压空白（含 SUBSTR 截断在标签中间的残片）。"""
    t = re.sub(r'<[^>]*>?', ' ', strip_body_noise(v))   # ⚠️ 先剥 <style>/<script> 块，否则块内 CSS/JS 文本会当正文留下
    t = re.sub(r'&nbsp;|&amp;|&lt;|&gt;|&#\d+;|&[a-z]+;', ' ', t)
    return re.sub(r'\s+', ' ', t).strip()[:n]


def _zc_item(r):
    ind = (r.get('industry') or '').strip()
    ph = (r.get('phase') or '').strip()
    loc = '/'.join([x for x in [(r.get('province') or '').strip(),
                                (r.get('city') or '').strip()] if x])
    tags = []
    if ind:
        tags.append({'v': ind, 'c': 'ind'})
    if ph:
        tags.append({'v': ph, 'c': ''})
    return {'id': str(r.get('project_id') or ''),
            'title': (r.get('project_name') or '').strip(),
            'site': loc,
            'date': (r.get('publish_date') or '')[:10],
            'tags': tags,
            'excerpt': _clean_txt(r.get('scale') or r.get('topic') or r.get('detail_address') or ''),
            'url': ''}


def _ccpc_item(r):
    ind = (r.get('industry') or '').strip()
    ph = (r.get('phase') or '').strip()
    tags = []
    if ind:
        tags.append({'v': ind, 'c': 'ind'})
    if ph:
        tags.append({'v': ph, 'c': ''})
    return {'id': str(r.get('id') or ''),
            'title': (r.get('project_name') or '').strip(),
            'site': (r.get('owner_company') or '').strip(),
            'date': (r.get('publish_date') or '')[:10],
            'tags': tags,
            'excerpt': _clean_txt(r.get('overview') or r.get('detail') or ''),
            'url': ''}


def _contact_item(r):
    tags = []
    _role = contact_role_display(r.get('role'))
    if _role:
        tags.append({'v': _role, 'c': 'ind'})
    _dp = ' · '.join([x for x in [(r.get('department') or '').strip(),
                                  (r.get('position') or '').strip()] if x])
    if _dp:
        tags.append({'v': _dp, 'c': ''})
    return {'id': str(r.get('id') or ''),
            'title': (r.get('company') or '').strip(),
            'site': ' · '.join([x for x in [(r.get('contact_name') or '').strip(),
                                            (r.get('phone') or '').strip()] if x]),
            'date': '',
            'tags': tags,
            'excerpt': _clean_txt(r.get('remarks') or r.get('address') or ''),
            'url': ''}


def _browse_generic(db_path, table, id_col, date_col, like_cols, row_fn,
                    id_is_int=False, cursor=None, limit=30, direction='next',
                    daterange='all', q='', with_total=False):
    """keyset 游标分页引擎（非 crawler 路由共用）。

    id_col    : 主键列（zc=TEXT project_id；ccpc/contact=INTEGER id）
    date_col  : 日期列；None = 该表无日期（contact，退化为 id 单键 keyset）
    like_cols : 关键词 LIKE 的列（多词之间 AND）
    row_fn    : 行 dict → 统一 item 形状
    """
    empty = {'items': [], 'next_cursor': '', 'prev_cursor': '', 'has_more': False,
             'total': None, 'count': 0}
    if not os.path.exists(db_path):
        return empty
    try:
        limit = max(1, min(int(limit or 30), 60))
    except Exception:
        limit = 30
    try:
        db = sqlite3.connect(db_path, timeout=10)
        db.row_factory = sqlite3.Row
    except Exception:
        return empty
    c = db.cursor()
    where, params, total = [], [], None

    days = LIST_DAYS.get(daterange)
    if days and date_col:
        # 与 /crawler 同一套脏格式防御：非标准日期不进时间窗口
        where.append("(%s GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' "
                     "AND SUBSTR(%s,1,10) >= date('now','localtime','-%d days'))"
                     % (date_col, date_col, days))

    for t in [x for x in re.split(r'[,\s]+', (q or '').strip()) if x][:4]:
        where.append('(' + ' OR '.join('%s LIKE ?' % col for col in like_cols) + ')')
        params.extend(['%' + t + '%'] * len(like_cols))

    cd, ci = (None, None)
    if cursor:
        cd, ci = _dec_curs(cursor)
    if ci not in (None, ''):
        try:
            ci_v = int(ci) if id_is_int else ci
        except Exception:
            ci_v = ci
        if date_col:
            # row-value 比较（SQLite >= 3.15）：与 /crawler 的 keyset 同一手法
            if direction == 'prev':
                where.append('(%s, %s) > (?, ?)' % (date_col, id_col))
                params.extend([cd, ci_v])
                order = '%s ASC, %s ASC' % (date_col, id_col)
            else:
                where.append('(%s, %s) < (?, ?)' % (date_col, id_col))
                params.extend([cd, ci_v])
                order = '%s DESC, %s DESC' % (date_col, id_col)
        else:
            where.append('%s %s ?' % (id_col, '>' if direction == 'prev' else '<'))
            params.append(ci_v)
            order = '%s %s' % (id_col, 'ASC' if direction == 'prev' else 'DESC')
    else:
        order = ('%s DESC, %s DESC' % (date_col, id_col)) if date_col else ('%s DESC' % id_col)
        if with_total:
            try:
                # 这三张表只有 1.6k~19k 行，带筛选算全量 COUNT 也是毫秒级 →
                # 带筛选时同样显示真实总数（crawler 那条路才有 43s 的问题）
                total = c.execute('SELECT COUNT(*) FROM %s%s' % (
                    table, (' WHERE ' + ' AND '.join(where)) if where else ''), params).fetchone()[0]
            except Exception:
                total = None

    rows = []
    try:
        c.execute('SELECT * FROM %s%s ORDER BY %s LIMIT ?'
                  % (table, (' WHERE ' + ' AND '.join(where)) if where else '', order),
                  params + [limit + 1])
        rows = [dict(x) for x in c.fetchall()]
    except Exception:
        rows = []
    try:
        db.close()
    except Exception:
        pass

    has_more = len(rows) > limit
    if has_more:
        rows = rows[:limit]
    if direction == 'prev':
        rows.reverse()
    items = [row_fn(r) for r in rows]
    nxt = _enc_curs(((rows[-1].get(date_col) or '')[:10] if date_col else ''),
                    rows[-1].get(id_col)) if rows else ''
    prv = _enc_curs(((rows[0].get(date_col) or '')[:10] if date_col else ''),
                    rows[0].get(id_col)) if rows else ''
    return {'items': items, 'next_cursor': nxt if has_more else '', 'prev_cursor': prv,
            'has_more': has_more, 'total': total, 'total_truncated': False,
            'count': len(items)}


def browse_route(route, cursor=None, limit=30, direction='next', daterange='all',
                 industry='', q='', kw='', with_total=False):
    """列表页 v2 数据入口（按路由分发）。crawler 走既有 browse_items，其余走 _browse_generic。"""
    if route == 'zc':
        return _browse_generic(ZC_DB_PATH, 'zc_projects', 'project_id', 'publish_date',
                               ('project_name', 'industry', 'province', 'city'), _zc_item,
                               cursor=cursor, limit=limit, direction=direction,
                               daterange=daterange, q=q, with_total=with_total)
    if route == 'ccpc':
        return _browse_generic(CONTACT_DB_PATH, 'cceup_projects', 'id', 'publish_date',
                               ('project_name', 'owner_company', 'industry', 'province', 'city'),
                               _ccpc_item, id_is_int=True, cursor=cursor, limit=limit,
                               direction=direction, daterange=daterange, q=q, with_total=with_total)
    if route == 'contact':
        return _browse_generic(CONTACT_ROUTE_DB, 'contacts', 'id', None,
                               ('company', 'contact_name', 'phone', 'landline', 'email',
                                'role', 'department', 'position'), _contact_item,
                               id_is_int=True, cursor=cursor, limit=limit, direction=direction,
                               daterange=daterange, q=q, with_total=with_total)
    return browse_items(cursor=cursor, limit=limit, direction=direction, daterange=daterange,
                        industry=industry, kw=kw, q=q, with_total=with_total)


def _qs_like(db_path, table, id_col, title_col, site_cols, date_col, q, limit):
    if not os.path.exists(db_path):
        return {'items': [], 'total': 0}
    cols = [title_col] + list(site_cols)
    w = ' OR '.join('%s LIKE ?' % col for col in cols)
    args = ['%' + q + '%'] * len(cols)
    order = ((date_col + ' DESC, ') if date_col else '') + (id_col + ' DESC')
    total, rows = 0, []
    try:
        db = sqlite3.connect(db_path, timeout=10)
        db.row_factory = sqlite3.Row
        total = db.execute('SELECT COUNT(*) FROM %s WHERE %s' % (table, w), args).fetchone()[0]
        rows = db.execute('SELECT * FROM %s WHERE %s ORDER BY %s LIMIT ?' % (table, w, order),
                          args + [limit]).fetchall()
        db.close()
    except Exception:
        total, rows = 0, []
    out = []
    for r in rows:
        site = ' · '.join([str(r[c]).strip() for c in site_cols
                           if r[c] is not None and str(r[c]).strip()])
        out.append({'id': str(r[id_col]) if r[id_col] is not None else '',
                    'title': str(r[title_col] or '').strip()[:120],
                    'site': site[:40],
                    'date': ((r[date_col] or '')[:10] if date_col else '')})
    return {'items': out, 'total': total}


def quicksearch_route(route, q, limit=12):
    """⌘K 快速搜索（按路由分发）。crawler 复用 list_quicksearch（实体/bigram/FTS 全语义）。"""
    q = (q or '').strip()
    if len(q) < 2:
        return {'items': [], 'total': 0}
    try:
        limit = max(1, min(int(limit or 12), 30))
    except Exception:
        limit = 12
    if route == 'zc':
        return _qs_like(ZC_DB_PATH, 'zc_projects', 'project_id', 'project_name',
                        ('industry', 'province', 'city'), 'publish_date', q, limit)
    if route == 'ccpc':
        return _qs_like(CONTACT_DB_PATH, 'cceup_projects', 'id', 'project_name',
                        ('owner_company', 'industry'), 'publish_date', q, limit)
    if route == 'contact':
        return _qs_like(CONTACT_ROUTE_DB, 'contacts', 'id', 'company',
                        ('contact_name', 'phone', 'landline'), None, q, limit)
    return list_quicksearch(q, limit)


def route_item_meta(route, iid):
    """/api/item 与 ⌘K 跳转需要的元数据：title/site/date/excerpt（按路由取库）。"""
    meta = {'title': '', 'site': '', 'date': '', 'excerpt': ''}
    try:
        if route == 'zc':
            db = sqlite3.connect(ZC_DB_PATH)
            db.row_factory = sqlite3.Row
            r = db.execute('SELECT project_name, industry, province, city, publish_date, '
                           'scale, topic FROM zc_projects WHERE project_id=?', (str(iid),)).fetchone()
            db.close()
            if r:
                loc = '/'.join([x for x in [(r['province'] or ''), (r['city'] or '')] if x])
                meta = {'title': r['project_name'] or '', 'site': loc or (r['industry'] or ''),
                        'date': (r['publish_date'] or '')[:10],
                        'excerpt': _clean_txt(r['scale'] or r['topic'] or '', 160)}
        elif route == 'ccpc':
            db = sqlite3.connect(CONTACT_DB_PATH)
            db.row_factory = sqlite3.Row
            r = db.execute('SELECT project_name, owner_company, publish_date, overview '
                           'FROM cceup_projects WHERE id=?', (int(iid),)).fetchone()
            db.close()
            if r:
                meta = {'title': r['project_name'] or '', 'site': r['owner_company'] or '',
                        'date': (r['publish_date'] or '')[:10],
                        'excerpt': _clean_txt(r['overview'] or '', 160)}
        elif route == 'contact':
            db = sqlite3.connect(CONTACT_ROUTE_DB)
            db.row_factory = sqlite3.Row
            r = db.execute('SELECT company, contact_name, phone, remarks, address '
                           'FROM contacts WHERE id=?', (int(iid),)).fetchone()
            db.close()
            if r:
                meta = {'title': r['company'] or '',
                        'site': ' · '.join([x for x in [(r['contact_name'] or ''),
                                                        (r['phone'] or '')] if x]),
                        'date': '', 'excerpt': _clean_txt(r['remarks'] or r['address'] or '', 160)}
        else:
            db = sqlite3.connect(SEARCH_DB)
            db.row_factory = sqlite3.Row
            r = db.execute('SELECT title, site_name, publish_date, '
                           "SUBSTR(COALESCE(summary,''),1,300) FROM gov_raw WHERE id=?",
                           (int(iid),)).fetchone()
            db.close()
            if r:
                meta = {'title': r[0] or '', 'site': r[1] or '', 'date': (r[2] or '')[:10],
                        'excerpt': _clean_txt(r[3] or '', 160)}
    except Exception:
        pass
    return meta


def render_ccpc_frag(pid):
    """中项网项目 → 抽屉详情片段（自包含：.dh/.ds 卡片风格，与 crawler/zc 抽屉一致）。

    ⚠️ 不动既有 handle_cceup_project_detail（/ccpc/project/{id} 整页仍在用），
       本函数是抽屉专用片段，避免改动线上页面引入回归。
    """
    try:
        db = sqlite3.connect(CONTACT_DB_PATH)
        db.row_factory = sqlite3.Row
        r = db.execute('SELECT * FROM cceup_projects WHERE id=?', (int(pid),)).fetchone()
        if not r:
            db.close()
            return ''
        p = dict(r)
        contacts = [dict(x) for x in db.execute(
            'SELECT company, contact_name, phone, landline, email, role, address '
            'FROM cceup_contacts WHERE project_id=? ORDER BY id', (int(pid),))]
        db.close()
    except Exception:
        return ''
    # 2026-09-11：标题单独一个 .dht 容器，与下方字段网格之间加细线分隔
    h = '<div class="dh"><div class="dht"><h1>' + esc(p.get('project_name') or '') + '</h1></div><div class="dg">'
    for lbl, key in [('项目编号', 'project_id'), ('行业', 'industry'), ('项目类型', 'project_type'),
                     ('项目阶段', 'phase'), ('建设性质', 'nature'), ('总投资', 'budget'),
                     ('资金性质', 'invest_nature'), ('资金来源', 'funding'), ('项目等级', 'grade'),
                     ('省份', 'province'), ('城市', 'city'), ('详细地址', 'project_address'),
                     ('业主单位', 'owner_company'), ('建筑面积', 'building_area'),
                     ('占地面积', 'land_area'), ('开工时间', 'start_time'),
                     ('竣工时间', 'end_time'), ('设备来源', 'equipment_source'),
                     ('发布日期', 'publish_date')]:
        v = str(p.get(key) or '').strip()
        if v:
            h += '<div><span class="lb">' + lbl + '：</span><span>' + esc(v) + '</span></div>'
    h += '</div></div>'
    for lbl, key in [('项目概况', 'overview'), ('项目详情', 'detail'), ('工艺流程', 'process_flow'),
                     ('设备清单', 'equipment_list'), ('采购要求', 'procurement_needs'),
                     ('阶段建设内容', 'phase_construction'), ('主要工作', 'key_works'),
                     ('项目进度', 'progress')]:
        v = str(p.get(key) or '').strip()
        if not v:
            continue
        v = strip_body_noise(v)                  # ⚠️ 剥 <style>/<script> 块（同上：注入会污染全页 CSS）
        if not v.strip():
            continue
        if body_is_html(v):
            h += '<div class="ds"><h3>' + lbl + '</h3><div class="ct-html">' + v + '</div></div>'
        else:
            h += '<div class="ds"><h3>' + lbl + '</h3><div class="ct">' + esc(v) + '</div></div>'
    if contacts:
        h += '<div class="ds"><h3>联系人（' + str(len(contacts)) + '）</h3>'
        for ct in contacts:
            h += '<div class="cc"><div class="cm">' + esc(ct.get('company') or '') + '</div><div class="fd">'
            for lbl, key in [('联系人', 'contact_name'), ('手机', 'phone'), ('座机', 'landline'),
                             ('邮箱', 'email'), ('地址', 'address')]:
                v = str(ct.get(key) or '').strip()
                if v:
                    h += '<span class="cl">' + lbl + '</span><span>' + esc(v) + '</span>'
            _role = contact_role_display(ct.get('role'))
            if _role:
                h += '<span class="cl">角色</span><span>' + esc(_role) + '</span>'
            h += '</div></div>'
        h += '</div>'
    return h


def render_contact_frag(cid):
    """联系人 → 抽屉详情片段：全字段卡片 + 关联项目（contact_projects 反查）。

    ⚠️ ccpc 来源的 contact_projects.project_key **恒为空串**（实测 11308/14947 行为空），
       只能按 project_name 反查 cceup_projects.id（实测 14947/14947 全部命中）；
       zc 来源的 project_key 就是 zc_projects.project_id，可直接跳。
    """
    try:
        cid_i = int(cid)
    except Exception:
        return ''
    try:
        db = sqlite3.connect(CONTACT_ROUTE_DB)
        db.row_factory = sqlite3.Row
        r = db.execute('SELECT * FROM contacts WHERE id=?', (cid_i,)).fetchone()
        if not r:
            db.close()
            return ''
        row = dict(r)
        rel = [dict(x) for x in db.execute(
            'SELECT source, project_key, project_name FROM contact_projects '
            'WHERE contact_id=? ORDER BY source, project_name LIMIT 60', (cid_i,))]
        db.close()
    except Exception:
        return ''
    # 2026-09-11：标题单独一个 .dht 容器，与下方字段网格之间加细线分隔
    h = '<div class="dh"><div class="dht"><h1>' + esc(row.get('company') or '') + '</h1></div><div class="dg">'
    _role = contact_role_display(row.get('role'))
    for lbl, v in [('联系人', row.get('contact_name')), ('角色', _role),
                   ('部门', row.get('department')), ('职务', row.get('position')),
                   ('手机', row.get('phone')), ('座机', row.get('landline')),
                   ('邮箱', row.get('email')), ('地址', row.get('address')),
                   ('备注', row.get('remarks'))]:
        v = str(v or '').strip()
        if v:
            h += '<div><span class="lb">' + lbl + '：</span><span>' + esc(v) + '</span></div>'
    h += '</div></div>'
    if rel:
        idmap = {}
        _names = sorted({(x.get('project_name') or '').strip() for x in rel
                         if (x.get('source') or '') == 'ccpc'
                         and (x.get('project_name') or '').strip()})
        if _names:
            try:
                kdb = sqlite3.connect(CONTACT_DB_PATH)
                kdb.row_factory = sqlite3.Row
                for rr in kdb.execute(
                        'SELECT id, project_name FROM cceup_projects WHERE project_name IN (%s)'
                        % ','.join('?' * len(_names)), _names):
                    idmap.setdefault((rr['project_name'] or '').strip(), rr['id'])
                kdb.close()
            except Exception:
                pass
        seen = set()
        h += '<div class="ds"><h3>关联项目（' + str(len(rel)) + '）</h3>'
        for x in rel:
            src = (x.get('source') or '').strip()
            key = (x.get('project_key') or '').strip()
            nm = (x.get('project_name') or '').strip() or key
            url = ''
            if src == 'zc' and key:
                url = '/zc/project/' + urllib.parse.quote(key)
            elif src == 'ccpc' and nm in idmap:
                url = '/ccpc/project/' + str(idmap[nm])
            if (url or nm) in seen:
                continue
            seen.add(url or nm)
            _tag = '中策' if src == 'zc' else ('中项网' if src == 'ccpc' else src)
            _nm_html = esc(nm)
            if url:
                _nm_html = ('<a href="' + esc(url) + '" target="_blank" rel="noopener" '
                            'style="color:#1a73e8;text-decoration:none">' + esc(nm) + '</a>')
            h += ('<div class="cc"><div class="cm">' + _nm_html + '</div>'
                  '<div class="fd"><span class="cl">来源</span><span>' + esc(_tag) + '</span></div></div>')
        h += '</div>'
    return h


# ══════════════════════════════════════════════════════════════════════════
#  主搜索页 ⌘K 快速搜索（2026-09-11）
#  与列表页 v2 同款交互：弹层 + 防抖 + ↑↓/Enter/Esc + 全局 ⌘K(⌘)/Ctrl+K。
#  数据源复用 /api/quicksearch（db=crawler → list_quicksearch → search() 全语义），
#  点击结果跳 /crawler/project/<id>。
#  注意：列表页的 .lp-cmk* 样式寄居在 LIST_UI_CSS 里，那份含 body{overflow:hidden}，
#  整块搬到主搜索页会把页面锁死，因此这里独立一份 CMK_CSS，只含弹层所需规则。
# ══════════════════════════════════════════════════════════════════════════
CMK_CSS = """<style>
.cmk{position:fixed;inset:0;background:rgba(20,24,32,.42);z-index:2000;display:none;align-items:flex-start;justify-content:center;padding-top:12vh}
.cmk.open{display:flex}
.cmk-box{width:620px;max-width:92vw;background:#fff;border-radius:14px;box-shadow:0 18px 60px rgba(0,0,0,.28);overflow:hidden}
.cmk-box input{width:100%;border:none;border-bottom:1px solid #eceff3;padding:15px 18px;font-size:15px;outline:none;font-family:inherit}
.cmk-res{max-height:52vh;overflow-y:auto}
.cmk-it{padding:10px 18px;cursor:pointer;border-bottom:1px solid #f4f6f8;font-size:13.5px}
.cmk-it.act,.cmk-it:hover{background:#f2f7ff}
.cmk-it .s{font-size:11.5px;color:#8a9099;margin-top:3px}
.cmk-empty{padding:12px 18px;color:#8a9099;font-size:13px}
.cmk-foot{padding:9px 18px;font-size:11.5px;color:#8a9099;border-top:1px solid #eceff3;display:flex;gap:14px;align-items:center}
.cmk-foot a{color:#1a73e8;text-decoration:none;margin-left:auto}
.cmk-tip{font-size:12px;color:#8a9099;margin-top:7px;text-align:center}
@media(max-width:760px){.cmk{padding-top:0}.cmk-box{width:100%;max-width:100vw;border-radius:0}.cmk-tip{display:none}}
</style>"""

CMK_HTML = """<div class="cmk" id="cmk"><div class="cmk-box">
<input id="cmk-in" placeholder="快速搜索：标题 / 公司全称 / 关键词…（Esc 关闭）" autocomplete="off">
<div class="cmk-res" id="cmk-res"></div>
<div class="cmk-foot"><span>↑↓ 选择</span><span>Enter 打开</span><span id="cmk-hint"></span>
<a href="/" id="cmk-all" onclick="return false">查看全部结果 →</a></div>
</div></div>"""

CMK_JS = """<script>
(function(){
  var r = document.getElementById('cmk'); if (!r) return;
  var inp = document.getElementById('cmk-in'), res = document.getElementById('cmk-res');
  var hint = document.getElementById('cmk-hint');
  var items = [], idx = -1, timer = null;
  function esc(s){ return String(s == null ? '' : s).replace(/[&<>"]/g, function(c){
    return ({'&':'&amp;','<':'&lt;','>':'&gt;','"':'&quot;'})[c] }) }
  function open(){ r.classList.add('open'); inp.value = ''; res.innerHTML = '';
    items = []; idx = -1; hint.textContent = ''; inp.focus() }
  function close(){ r.classList.remove('open') }
  function render(list, total){
    items = list || []; idx = items.length ? 0 : -1;
    var h = items.map(function(it, i){
      return '<div class="cmk-it' + (i === 0 ? ' act' : '') + '" data-i="' + i + '" data-id="' + esc(it.id) + '">'
        + esc(it.title) + '<div class="s">' + esc(it.site) + ' · ' + esc(it.date) + '</div></div>' }).join('');
    if (!h) h = '<div class="cmk-empty">无匹配结果</div>';
    res.innerHTML = h;
    hint.textContent = total ? ('共 ' + total + ' 条匹配，前 ' + items.length + ' 条') : '';
  }
  function go(id){ if (id) location.href = '/crawler/project/' + encodeURIComponent(id) }
  inp.addEventListener('input', function(){
    var v = this.value; clearTimeout(timer);
    if (v.trim().length < 2){ render([], 0); return }
    timer = setTimeout(function(){
      fetch('/api/quicksearch?db=crawler&q=' + encodeURIComponent(v))
        .then(function(x){ return x.json() })
        .then(function(d){ if (d && d.ok !== false) render(d.items, d.total) })
        .catch(function(){ render([], 0) });
    }, 160);
  });
  res.addEventListener('click', function(e){
    var el = e.target.closest && e.target.closest('.cmk-it');
    if (!el || !el.dataset.id) return;
    close(); go(el.dataset.id);
  });
  inp.addEventListener('keydown', function(e){
    if (e.key === 'ArrowDown' || e.key === 'ArrowUp'){
      e.preventDefault(); if (!items.length) return;
      idx += (e.key === 'ArrowDown' ? 1 : -1);
      if (idx < 0) idx = 0;
      if (idx > items.length - 1) idx = items.length - 1;
      var els = res.querySelectorAll('.cmk-it');
      Array.prototype.forEach.call(els, function(el, i){ el.classList.toggle('act', i === idx) });
      if (els[idx]) els[idx].scrollIntoView({block:'nearest'});
    } else if (e.key === 'Enter'){
      e.preventDefault(); var it = items[idx]; close(); if (it) go(it.id);
    } else if (e.key === 'Escape'){ e.preventDefault(); close() }
  });
  r.addEventListener('click', function(e){ if (e.target === this) close() });
  document.getElementById('cmk-all').addEventListener('click', function(e){
    e.preventDefault(); var v = inp.value.trim();
    if (v) location.href = '/?q=' + encodeURIComponent(v);
  });
  document.addEventListener('keydown', function(e){
    if ((e.metaKey || e.ctrlKey) && (e.key === 'k' || e.key === 'K')){ e.preventDefault(); open(); return }
    if (e.key === 'Escape' && r.classList.contains('open')) close();
  });
})();
</script>"""

CMK_TAIL = CMK_HTML + CMK_JS


# ══════════════════════════════════════════════════════════════════════════
#  /tools 实用工具市场（2026-09-11 新增）
#
#  资源来源：坚果云「Tools」目录 → 由 sync_tools_to_server.sh 上传到 TOOLS_DIR。
#  目录结构：TOOLS_DIR/<工具目录或单文件>；可选 TOOLS_DIR/tools.json 人工策展
#            （name/version/summary/category/icon/platform/tags/homepage/entry）；
#            未策展条目按目录名自动推断（'Xxx_v1.2.3' → 名称 Xxx / 版本 1.2.3）。
#  打包下载：目录型工具首次下载时压成 TOOLS_DIR/.zips/<id>.zip 并缓存
#            （源目录 mtime 更新则重压）；单文件型直接原样下载。
#  设计：卡片网格的「软件市场」形态，配色/圆角/字体沿用全站设计令牌。
# ══════════════════════════════════════════════════════════════════════════
import os as _os_tools
import re as _re_tools
import json as _json_tools
import time as _time_tools
import shutil as _shutil_tools

TOOLS_DIR = '/mnt/data/tools'
TOOLS_MANIFEST = '/mnt/data/tools/tools.json'
TOOLS_ZIP_DIR = '/mnt/data/tools/.zips'
_TOOLS_SIZE_CACHE = {}

TOOLS_CATEGORIES = [
    {'id': 'all',   'cn': '全部',     'icon': '🧰', 'tint': '#1a73e8'},
    {'id': 'ocr',   'cn': '文字识别', 'icon': '🔍', 'tint': '#1a73e8'},
    {'id': 'doc',   'cn': '文档处理', 'icon': '📄', 'tint': '#7b3fa0'},
    {'id': 'data',  'cn': '数据处理', 'icon': '📊', 'tint': '#1e7d45'},
    {'id': 'net',   'cn': '网络工具', 'icon': '🌐', 'tint': '#b26a00'},
    {'id': 'media', 'cn': '影音图像', 'icon': '🎬', 'tint': '#d93025'},
    {'id': 'dev',   'cn': '开发辅助', 'icon': '⚙️', 'tint': '#3c4043'},
    {'id': 'other', 'cn': '其他',     'icon': '📦', 'tint': '#5f6368'},
]


def _tools_cat(cid):
    for c in TOOLS_CATEGORIES:
        if c['id'] == cid:
            return c
    return TOOLS_CATEGORIES[-1]


def tools_human(n):
    try:
        n = float(n or 0)
    except Exception:
        return '—'
    for u in ('B', 'KB', 'MB', 'GB'):
        if n < 1024:
            return ('%d B' % n) if u == 'B' else ('%.1f %s' % (n, u))
        n /= 1024.0
    return '%.1f TB' % n


def tools_dir_size(path):
    """(总字节, 文件数)。带 mtime 缓存 —— 120MB 目录树不必每次请求都 walk。"""
    try:
        st = _os_tools.stat(path)
    except OSError:
        return 0, 0
    hit = _TOOLS_SIZE_CACHE.get(path)
    if hit and hit[0] == int(st.st_mtime):
        return hit[1], hit[2]
    total, cnt = 0, 0
    if _os_tools.path.isdir(path):
        for root, dirs, files in _os_tools.walk(path):
            dirs[:] = [d for d in dirs if not d.startswith('.')]
            for f in files:
                try:
                    total += _os_tools.path.getsize(_os_tools.path.join(root, f))
                    cnt += 1
                except OSError:
                    pass
    else:
        total, cnt = st.st_size, 1
    _TOOLS_SIZE_CACHE[path] = (int(st.st_mtime), total, cnt)
    return total, cnt


def tools_manifest():
    try:
        with open(TOOLS_MANIFEST, encoding='utf-8') as f:
            d = _json_tools.load(f)
        return d if isinstance(d, dict) else {}
    except Exception:
        return {}


# 可识别的「归档/安装/脚本」后缀：推断名称与版本前先剥掉，
# 否则 '测试数据包_v1.2.zip' 的正则被 '.zip' 挡住，名称会带着扩展名、版本提不出来。
_TOOLS_FILE_EXT = ('.tar.gz', '.tar.bz2', '.tar.xz', '.tar', '.zip', '.7z', '.rar', '.gz',
                   '.exe', '.msi', '.dmg', '.pkg', '.apk', '.iso', '.jar', '.appimage',
                   '.sh', '.command', '.bat', '.ps1', '.py')


def _tools_strip_ext(nm):
    low = nm.lower()
    for e in _TOOLS_FILE_EXT:
        if low.endswith(e):
            return nm[:len(nm) - len(e)]
    return nm


def _tools_ext(nm):
    low = nm.lower()
    for e in _TOOLS_FILE_EXT:
        if low.endswith(e):
            return e.lstrip('.').upper()
    dot = nm.rfind('.')
    return nm[dot + 1:].upper()[:6] if dot > 0 else ''


def _tools_ascii_name(fname):
    """Content-Disposition 的 ASCII 兜底名。中文名剥完 ASCII 只剩 '.sh' / '_v1.2.zip'
    这类残片（老客户端会照存），必须换成 tool-download.<ext>。"""
    a = fname.encode('ascii', 'ignore').decode().strip()
    if (not a) or (not a[0].isalnum()):
        ext = _os_tools.path.splitext(fname)[1]
        return 'tool-download' + ext
    return a


def _tools_guess(nm):
    """'Umi-OCR_Paddle_v2.1.5' → ('Umi-OCR_Paddle', '2.1.5')；先剥扩展名。"""
    nm = _tools_strip_ext(nm)
    m = _re_tools.match(r'^(.+?)[ _\-]+[vV]?(\d+(?:\.\d+)*)$', nm)
    return (m.group(1), m.group(2)) if m else (nm, '')


def _tools_top_files(path, limit=14):
    out = []
    try:
        if _os_tools.path.isfile(path):
            return [{'n': _os_tools.path.basename(path), 's': tools_human(_os_tools.path.getsize(path)), 'd': False}]
        for nm in sorted(_os_tools.listdir(path)):
            if nm.startswith('.'):
                continue
            fp = _os_tools.path.join(path, nm)
            try:
                if _os_tools.path.isdir(fp):
                    sz, c = tools_dir_size(fp)
                    out.append({'n': nm + '/', 's': tools_human(sz), 'd': True})
                else:
                    out.append({'n': nm, 's': tools_human(_os_tools.path.getsize(fp)), 'd': False})
            except OSError:
                continue
            if len(out) >= limit:
                break
    except OSError:
        pass
    return out


def tools_safe_id(tid):
    """校验工具 id 只能是一级目录/文件名（防目录穿越）"""
    tid = (tid or '').strip()
    if (not tid) or tid.startswith('.') or ('/' in tid) or ('\\' in tid) or ('..' in tid):
        return ''
    if tid != _os_tools.path.basename(tid):
        return ''
    return tid if _os_tools.path.exists(_os_tools.path.join(TOOLS_DIR, tid)) else ''


def scan_tools():
    """扫描 TOOLS_DIR → 卡片数据列表"""
    man = tools_manifest()
    curated = man.get('tools') or {}
    out = []
    try:
        names = sorted(_os_tools.listdir(TOOLS_DIR))
    except OSError:
        return out
    for nm in names:
        if nm.startswith('.') or nm == 'tools.json':
            continue
        p = _os_tools.path.join(TOOLS_DIR, nm)
        if not (_os_tools.path.isdir(p) or _os_tools.path.isfile(p)):
            continue
        c = curated.get(nm) or {}
        gname, gver = _tools_guess(nm)
        size, cnt = tools_dir_size(p)
        isdir = _os_tools.path.isdir(p)
        entry = c.get('entry') or ''
        if not entry and isdir:
            try:
                for f in sorted(_os_tools.listdir(p)):
                    if f.lower().endswith(_TOOLS_FILE_EXT):
                        entry = f
                        break
            except OSError:
                pass
        zp = _os_tools.path.join(TOOLS_ZIP_DIR, nm + '.zip')
        cid = c.get('category') or 'other'
        cat = _tools_cat(cid)
        tops = _tools_top_files(p)
        ext = '' if isdir else _tools_ext(nm)
        # 未策展条目也要有摘要，否则卡片空一半（用户 2026-09-11 实测反馈）
        summary = c.get('summary') or ''
        if not summary:
            if isdir:
                names = [_os_tools.path.basename(x['n'].rstrip('/')) for x in tops[:3]]
                summary = ('包含 %d 项：%s%s' % (cnt, '、'.join(names), ' 等' if cnt > len(names) else '')
                           if names else '空目录')
            else:
                summary = '%s 文件 · %s' % (ext or '未知类型', tools_human(size))
        tags = list(c.get('tags') or [])
        if ext and ext not in tags:
            tags.insert(0, ext)   # 单文件自动带类型标签（ZIP / SH / EXE / APK …）
        out.append({
            'id': nm,
            'name': c.get('name') or gname,
            'version': c.get('version') or gver,
            'summary': summary,
            'category': cid,
            'cat_cn': cat['cn'],
            'platform': c.get('platform') or ('文件夹' if isdir else ((ext + ' 单文件') if ext else '单文件')),
            'tags': tags,
            'icon': c.get('icon') or cat['icon'],
            'tint': cat['tint'],
            'homepage': c.get('homepage') or '',
            'entry': entry,
            'is_dir': isdir,
            'size': size,
            'size_h': tools_human(size),
            'files': cnt,
            'updated': _time_tools.strftime('%Y-%m-%d', _time_tools.localtime(_os_tools.path.getmtime(p))),
            'zip_ready': _os_tools.path.isfile(zp),
            'zip_h': tools_human(_os_tools.path.getsize(zp)) if _os_tools.path.isfile(zp) else '',
            'top_files': tops,
            'pinned': bool(c.get('pinned')),
        })
    out.sort(key=lambda t: (not t['pinned'], t['name'].lower()))
    return out


def tools_build_zip(tid):
    """→ (ok, 可下载文件路径, 错误)。目录型压成 .zips/<id>.zip（mtime 旧则重压并缓存）。"""
    safe = tools_safe_id(tid)
    if not safe:
        return False, '', '工具不存在'
    src = _os_tools.path.join(TOOLS_DIR, safe)
    if _os_tools.path.isfile(src):
        return True, src, ''
    try:
        _os_tools.makedirs(TOOLS_ZIP_DIR, exist_ok=True)
        zp = _os_tools.path.join(TOOLS_ZIP_DIR, safe + '.zip')
        if _os_tools.path.isfile(zp) and _os_tools.path.getmtime(zp) >= _os_tools.path.getmtime(src):
            return True, zp, ''
        tmpbase = zp + '.part'
        for stale in (tmpbase + '.zip',):
            try:
                _os_tools.remove(stale)
            except OSError:
                pass
        _shutil_tools.make_archive(tmpbase, 'zip', root_dir=src)
        _os_tools.replace(tmpbase + '.zip', zp)
        return True, zp, ''
    except Exception as e:
        return False, '', '打包失败: %s' % e


TOOLS_CSS = """<style>
*{margin:0;padding:0;box-sizing:border-box}
body{background:#f1f3f5;color:#202124;font-family:-apple-system,BlinkMacSystemFont,"Microsoft YaHei","PingFang SC",sans-serif;-webkit-font-smoothing:antialiased}
a{text-decoration:none}
.tw{max-width:1160px;margin:0 auto;padding:16px 20px 64px}
.tnav{display:flex;align-items:center;gap:14px;flex-wrap:wrap;margin-bottom:20px}
.tlogo{font-weight:700;color:#1a73e8;font-size:17px;letter-spacing:-.4px}
.tlinks{display:flex;gap:4px;flex-wrap:wrap}
.tlinks a{font-size:13px;color:#5f6368;padding:5px 11px;border-radius:16px}
.tlinks a:hover{background:#fff;color:#1a73e8}
.tlinks a.act{background:#e8f0fe;color:#1a73e8;font-weight:600}
.thero{background:linear-gradient(135deg,#1a73e8 0%,#3f8bf0 55%,#5aa0f5 100%);border-radius:16px;padding:26px 28px;color:#fff;box-shadow:0 6px 22px rgba(26,115,232,.22);position:relative;overflow:hidden}
.thero:after{content:"";position:absolute;right:-60px;top:-70px;width:230px;height:230px;border-radius:50%;background:rgba(255,255,255,.10)}
.thero:before{content:"";position:absolute;right:56px;bottom:-96px;width:170px;height:170px;border-radius:50%;background:rgba(255,255,255,.08)}
.thero-t{font-size:25px;font-weight:700;letter-spacing:-.5px;position:relative}
.thero-s{font-size:13.5px;opacity:.92;margin-top:7px;position:relative}
.thero-st{display:flex;gap:22px;margin-top:18px;flex-wrap:wrap;position:relative}
.thero-st div{font-size:12.5px;opacity:.95}
.thero-st b{display:block;font-size:19px;font-weight:700;margin-bottom:2px}
.tbar{display:flex;align-items:center;gap:10px;flex-wrap:wrap;margin:20px 0 14px}
.tq{flex:1;min-width:220px;position:relative}
.tq input{width:100%;padding:11px 14px 11px 38px;border:1px solid #dfe3e8;border-radius:11px;font-size:14px;outline:none;background:#fff;font-family:inherit}
.tq input:focus{border-color:#1a73e8;box-shadow:0 0 0 3px rgba(26,115,232,.12)}
.tq span{position:absolute;left:13px;top:50%;transform:translateY(-50%);font-size:14px;opacity:.45}
.tchips{display:flex;gap:6px;flex-wrap:wrap}
.tchip{padding:7px 13px;border:1px solid #dfe3e8;border-radius:999px;background:#fff;font-size:12.5px;color:#3c4043;cursor:pointer;white-space:nowrap;font-family:inherit}
.tchip:hover{border-color:#1a73e8;color:#1a73e8}
.tchip.on{background:#1a73e8;border-color:#1a73e8;color:#fff;font-weight:600}
.tcount{font-size:12.5px;color:#7a8089;margin-left:auto;white-space:nowrap}
.tgrid{display:grid;grid-template-columns:repeat(auto-fill,minmax(318px,1fr));gap:14px}
.tcard{background:#fff;border:1px solid #e9ecef;border-radius:14px;padding:18px;display:flex;flex-direction:column;gap:11px;transition:box-shadow .18s,transform .18s,border-color .18s}
.tcard:hover{box-shadow:0 10px 26px rgba(20,30,50,.10);transform:translateY(-2px);border-color:#dbe4f0}
.thead{display:flex;gap:12px;align-items:flex-start}
.ticon{width:50px;height:50px;border-radius:13px;display:flex;align-items:center;justify-content:center;font-size:24px;flex:0 0 auto;background:color-mix(in srgb,var(--tint) 12%,#fff);border:1px solid color-mix(in srgb,var(--tint) 24%,#fff)}
.ttitle{min-width:0;flex:1}
.tname{font-size:15.5px;font-weight:650;line-height:1.3;word-break:break-word}
.tver{display:inline-block;font-size:11px;font-weight:600;color:#1a73e8;background:#e8f0fe;border-radius:6px;padding:2px 7px;margin-left:6px;vertical-align:1px}
.tsub{font-size:12px;color:#7a8089;margin-top:4px}
.tsum{font-size:13px;color:#3c4043;line-height:1.62;flex:1}
.tmeta{display:flex;gap:12px;flex-wrap:wrap;font-size:11.5px;color:#7a8089;padding-top:2px}
.ttags{display:flex;gap:5px;flex-wrap:wrap}
.ttag{font-size:11px;color:#5f6368;background:#f1f3f5;border-radius:6px;padding:2px 8px}
.tacts{display:flex;gap:8px;align-items:center;margin-top:2px}
.tbtn{display:inline-flex;align-items:center;justify-content:center;gap:6px;padding:9px 16px;border-radius:10px;font-size:13px;font-weight:600;cursor:pointer;border:1px solid transparent;font-family:inherit;white-space:nowrap}
.tbtn.p{background:#1a73e8;color:#fff}
.tbtn.p:hover{background:#1557b0}
.tbtn.g{background:#fff;color:#3c4043;border-color:#dfe3e8}
.tbtn.g:hover{border-color:#1a73e8;color:#1a73e8}
.tdet{flex:1;min-width:0}
.tdet summary{list-style:none;cursor:pointer;font-size:12.5px;color:#5f6368;padding:9px 12px;border:1px solid #e9ecef;border-radius:10px;text-align:center}
.tdet summary::-webkit-details-marker{display:none}
.tdet summary:hover{border-color:#1a73e8;color:#1a73e8}
.tdet[open] summary{border-color:#1a73e8;color:#1a73e8;border-bottom-left-radius:0;border-bottom-right-radius:0}
.tfl{border:1px solid #e9ecef;border-top:none;border-radius:0 0 10px 10px;max-height:210px;overflow-y:auto;font-size:12px}
.tfl div{display:flex;gap:8px;padding:6px 12px;border-bottom:1px solid #f4f6f8}
.tfl div:last-child{border-bottom:none}
.tfl .fn{flex:1;min-width:0;overflow:hidden;text-overflow:ellipsis;white-space:nowrap;color:#3c4043}
.tfl .fs{color:#9aa0a6;flex:0 0 auto}
.tfl .fd{color:#1a73e8}
.tfoot{margin-top:26px;font-size:11.5px;color:#9aa0a6;text-align:center;line-height:1.9}
.tempty{background:#fff;border:1px dashed #dfe3e8;border-radius:14px;padding:46px 20px;text-align:center;color:#7a8089;font-size:14px;line-height:2}
.tempty code{background:#f1f3f5;border-radius:5px;padding:2px 7px;font-size:12.5px;color:#3c4043}
@media(max-width:760px){
  .tw{padding:12px 12px 48px}
  .thero{padding:20px 18px;border-radius:14px}
  .thero-t{font-size:21px}
  .tgrid{grid-template-columns:1fr;gap:12px}
  .tcount{margin-left:0;width:100%}
  .tq{min-width:100%}
  .tbtn{padding:9px 14px}
}
</style>"""


def render_tools_page(username=''):
    tools = scan_tools()
    total_size = sum(t['size'] for t in tools)
    cats_present = []
    for c in TOOLS_CATEGORIES:
        if c['id'] == 'all':
            continue
        if any(t['category'] == c['id'] for t in tools):
            cats_present.append(c)

    h = ['<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8">',
         '<meta name="viewport" content="width=device-width,initial-scale=1">',
         '<title>工具市场 · BJIIR项目库</title>',
         '<link rel="icon" href="data:,">',
         TOOLS_CSS, '</head><body><div class="tw">']

    # 导航
    h.append('<div class="tnav"><a class="tlogo" href="/">BJIIR</a><div class="tlinks">')
    # 这里**刻意不列**「工具市场」：nav_right_html() 已在右上角放了一个（且全站可见），
    # 本页再列一次会出现两个「工具市场」（2026-09-11 浏览器实测）。
    for _hid, href, label in [('crawler', '/crawler/', '环评公示'), ('zc', '/zc/', '中策大数据'),
                              ('ccpc', '/ccpc', '中项网'), ('contact', '/contact', '联系人DB'),
                              ('originals', '/originals/', '原始数据')]:
        h.append('<a href="' + href + '">' + label + '</a>')
    h.append('</div>' + nav_right_html(username) + '</div>')

    # Hero
    h.append('<div class="thero"><div class="thero-t">🧰 工具市场</div>'
             '<div class="thero-s">内部实用工具，点开即下 · 目录与坚果云 Tools 同步</div>'
             '<div class="thero-st">'
             '<div><b>' + str(len(tools)) + '</b>个工具</div>'
             '<div><b>' + tools_human(total_size) + '</b>合计体积</div>'
             '<div><b>' + str(sum(t['files'] for t in tools)) + '</b>个文件</div>'
             '</div></div>')

    if not tools:
        h.append('<div class="tempty" style="margin-top:20px">🧰 工具目录还是空的<br>'
                 '把工具放进服务器的 <code>' + TOOLS_DIR + '</code>，'
                 '或在本机跑 <code>sync_tools_to_server.sh</code> 从坚果云 Tools 目录同步<br>'
                 '（可选：加一份 <code>tools.json</code> 写简介/分类/版本，页面会自动读取）</div>')

    if tools:
        # 工具条：搜索 + 分类
        h.append('<div class="tbar"><div class="tq"><span>🔍</span>'
                 '<input id="tq" placeholder="搜索工具名 / 简介 / 标签…" autocomplete="off"></div>'
                 '<div class="tchips" id="tchips">')
        h.append('<button class="tchip on" data-cat="all" type="button">🧰 全部</button>')
        for c in cats_present:
            h.append('<button class="tchip" data-cat="' + c['id'] + '" type="button">'
                     + c['icon'] + ' ' + c['cn'] + '</button>')
        h.append('</div><div class="tcount" id="tcount">共 ' + str(len(tools)) + ' 个工具</div></div>')

        # 卡片
        h.append('<div class="tgrid" id="tgrid">')
        for t in tools:
            hay = ' '.join([t['name'], t['summary'], t['id'], t['cat_cn'], t['platform']] + list(t['tags']))
            h.append('<article class="tcard" data-cat="' + t['category'] + '" data-hay="'
                     + esc(hay.lower()) + '">')
            h.append('<div class="thead"><div class="ticon" style="--tint:' + t['tint'] + '">'
                     + esc(t['icon']) + '</div><div class="ttitle">'
                     '<div class="tname">' + esc(t['name'])
                     + ('<span class="tver">v' + esc(t['version']) + '</span>' if t['version'] else '')
                     + '</div><div class="tsub">' + esc(t['cat_cn']) + ' · ' + esc(t['platform'])
                     + ((' · 入口 ' + esc(t['entry'])) if t['entry'] else '') + '</div></div></div>')
            if t['summary']:
                h.append('<div class="tsum">' + esc(t['summary']) + '</div>')
            h.append('<div class="tmeta"><span>📦 ' + esc(t['size_h']) + '</span>'
                     '<span>🗂 ' + str(t['files']) + ' 个文件</span>'
                     '<span>🕒 ' + esc(t['updated']) + '</span>'
                     + ('<span>🗜 ' + esc(t['zip_h']) + '</span>' if t['zip_h'] else '')
                     + '</div>')
            if t['tags']:
                h.append('<div class="ttags">'
                         + ''.join('<span class="ttag">' + esc(x) + '</span>' for x in t['tags'])
                         + '</div>')
            h.append('<div class="tacts">'
                     '<a class="tbtn p" href="/tools/download/' + urllib.parse.quote(t['id']) + '">⬇ 下载'
                     + ('（打包 ' + esc(t['zip_h'] or t['size_h']) + '）' if False else '') + '</a>')
            if t['homepage']:
                h.append('<a class="tbtn g" href="' + esc(t['homepage']) + '" target="_blank" rel="noopener">官网</a>')
            h.append('</div>')
            if t['top_files']:
                h.append('<details class="tdet"><summary>包含文件（前 ' + str(len(t['top_files'])) + ' 项）</summary>'
                         '<div class="tfl">'
                         + ''.join('<div><span class="fn' + (' fd' if f['d'] else '') + '">' + esc(f['n'])
                                   + '</span><span class="fs">' + esc(f['s']) + '</span></div>'
                                   for f in t['top_files'])
                         + '</div></details>')
            h.append('</article>')
        h.append('</div>')
        h.append('<div class="tempty" id="tempty" style="display:none">🔍 没有匹配的工具，换个关键词试试</div>')

    h.append('<div class="tfoot">资源目录 <code style="color:#7a8089">' + TOOLS_DIR + '</code>'
             ' · 坚果云 Tools 同步 · 打包下载首次约 5~15s，之后走缓存</div>')
    h.append('</div>')

    if tools:
        h.append("""<script>
(function(){
  var q = document.getElementById('tq'), grid = document.getElementById('tgrid'),
      chips = document.getElementById('tchips'), cnt = document.getElementById('tcount'),
      empty = document.getElementById('tempty');
  var cat = 'all', cards = [].slice.call(grid.querySelectorAll('.tcard'));
  function apply(){
    var kw = (q.value || '').trim().toLowerCase(), shown = 0;
    cards.forEach(function(c){
      var ok = (cat === 'all' || c.dataset.cat === cat) && (!kw || c.dataset.hay.indexOf(kw) >= 0);
      c.style.display = ok ? '' : 'none';
      if (ok) shown++;
    });
    cnt.textContent = kw || cat !== 'all' ? ('匹配 ' + shown + ' / ' + cards.length + ' 个工具') : ('共 ' + cards.length + ' 个工具');
    empty.style.display = shown ? 'none' : '';
  }
  q.addEventListener('input', apply);
  chips.addEventListener('click', function(e){
    var b = e.target.closest && e.target.closest('.tchip'); if (!b) return;
    cat = b.dataset.cat;
    [].forEach.call(chips.querySelectorAll('.tchip'), function(x){ x.classList.toggle('on', x === b) });
    apply();
  });
})();
</script>""")
    h.append('</body></html>')
    return ''.join(h)


def render_cceup_detail_body(p, contacts):
    """中项网(CCPC)项目 → 详情正文卡片。

    2026-09-11：由老设计（.detail-wrap/.detail-title/.detail-card/.detail-grid/
    .label/.value/.detail-text）统一到 v2 的 .dh/.dht/.dg/.lb/.ds/.ct/.cc，
    与列表页右侧抽屉（render_ccpc_frag）及 crawler/zc 详情页共用同一套 DETAIL_CSS。
    字段与老版逐一对齐：基本信息 15 项 + 工期与规模 + 结构与配套 + 8 个文本段 +
    预测说明/项目详情 + 业主 + 联系人。不含 <html>/<style>/页壳。
    """
    h = '<div class="dh"><div class="dht"><h1>' + esc(p.get('project_name') or '') + '</h1></div>'

    def _grid(pairs):
        out = ''
        for lbl, key in pairs:
            v = p.get(key, '')
            v = '' if v is None else str(v)
            if v.strip() and v not in ('一一', '——'):
                out += '<div><span class="lb">' + lbl + '：</span><span>' + esc(v) + '</span></div>'
        return out

    h += '<div class="dg">' + _grid([
        ('项目编号', 'project_id'), ('项目类型', 'project_type'), ('项目阶段', 'phase'),
        ('投资金额', 'budget'), ('行业', 'industry'), ('细分行业', 'field_type'),
        ('工程类型', 'nature'), ('投资性质', 'invest_nature'), ('资金情况', 'funding'),
        ('项目等级', 'grade'), ('省份', 'province'), ('城市', 'city'),
        ('地址', 'project_address'), ('发布日期', 'publish_date'), ('版本', 'version'),
    ]) + '</div></div>'

    # 工期与规模 / 结构与配套：各成一张 .ds 卡片（内层复用 .dg 两列网格）
    for _title, _pairs in [
        ('工期与规模', [('开工时间', 'start_time'), ('工期', 'end_time'),
                        ('建筑面积', 'building_area'), ('占地面积', 'land_area'),
                        ('设备来源', 'equipment_source')]),
        ('结构与配套', [('钢结构', 'steel_structure'), ('建筑层数', 'building_floors'),
                        ('供暖方式', 'heating_method'), ('外墙材料', 'wall_material'),
                        ('装修', 'decoration'), ('有无空调', 'has_ac'),
                        ('有无立体停车位', 'has_parking'), ('有无电梯', 'has_elevator')]),
    ]:
        _inner = _grid(_pairs)
        if _inner:
            h += '<div class="ds"><h3>' + _title + '</h3><div class="dg">' + _inner + '</div></div>'

    # 文本段（.ct 自带 white-space:pre-line，换行保留）
    for lbl, key in [('项目概况', 'overview'), ('工艺流程', 'process_flow'),
                     ('采购设备', 'procurement_equipment'), ('采购需求', 'procurement_needs'),
                     ('项目进展', 'progress'), ('设备清单', 'equipment_list'),
                     ('商机', 'opportunity'), ('重点工作', 'key_works')]:
        v = p.get(key, '')
        v = '' if v is None else str(v)
        if v.strip() and v not in ('一一', '——'):
            h += '<div class="ds"><h3>' + lbl + '</h3><div class="ct">' + esc(v) + '</div></div>'

    # detail 字段：带「：」的行拆成键值（→ 预测说明），其余保留为正文（→ 项目详情）
    _dv = p.get('detail', '')
    _dv = '' if _dv is None else str(_dv)
    if _dv.strip() and _dv not in ('一一', '——'):
        table_rows, text_parts = [], []
        for line in _dv.split('\n'):
            line = line.strip()
            if not line:
                continue
            if '：' in line:
                k, _, v = line.partition('：')
                k, v = k.strip(), v.strip()
                if k and v and k not in ('【所需材料设备】', '【项目进展】', '【项目详情】', '版权所有'):
                    table_rows.append((k, v))
                    continue
            text_parts.append(line)
        if table_rows:
            h += ('<div class="ds"><h3>预测说明</h3><div class="ct">'
                  + '\n'.join(esc(k) + '：' + esc(v) for k, v in table_rows) + '</div></div>')
        if text_parts:
            h += ('<div class="ds"><h3>项目详情</h3><div class="ct">'
                  + esc('\n'.join(text_parts)) + '</div></div>')

    if p.get('owner_company'):
        h += '<div class="ds"><h3>业主</h3><div class="ct">' + esc(p['owner_company']) + '</div></div>'

    if contacts:
        h += '<div class="ds"><h3>联系人（' + str(len(contacts)) + '）</h3>'
        for c in contacts:
            h += ('<div class="cc"><div class="cm">' + esc(c.get('company') or '') + '</div>'
                  '<div class="fd">')
            for lbl, key in [('姓名', 'contact_name'), ('电话', 'phone'),
                             ('邮箱', 'email'), ('地址', 'address')]:
                v = '' if c.get(key) is None else str(c.get(key)).strip()
                if v:
                    h += '<span class="cl">' + lbl + '</span><span>' + esc(v) + '</span>'
            h += '</div></div>'
        h += '</div>'

    return h



class SearchHandler(http.server.BaseHTTPRequestHandler):

    def _proxy_tampermonkey(self, raw_path):
        """反向代理 /tm/* → 127.0.0.1:8010 (Tampermonkey script center)"""
        import http.client
        # 去掉 /tm 前缀，转发到 8010
        target = raw_path
        if target.startswith('/tm'):
            target = target[len('/tm'):] or '/'
        try:
            # 透传 Cookie（tm_lang=zh|en 语言切换依赖）与 X-Forwarded-Prefix
            _hdrs = {'X-Forwarded-Prefix': '/tm'}
            _ck = self.headers.get('Cookie', '')
            if _ck:
                _hdrs['Cookie'] = _ck
            conn = http.client.HTTPConnection('127.0.0.1', 8010, timeout=15)
            conn.request('GET', target, headers=_hdrs)
            resp = conn.getresponse()
            body = resp.read()
            self.send_response(resp.status)
            for k, v in resp.getheaders():
                if k.lower() not in ('content-length', 'transfer-encoding', 'connection'):
                    self.send_header(k, v)
            self.send_header('Content-Length', str(len(body)))
            self.end_headers()
            self.wfile.write(body)
            conn.close()
        except Exception as e:
            try:
                self.send_response(502)
                self.send_header('Content-Type', 'text/plain; charset=utf-8')
                self.send_header('Content-Length', str(len(str(e)) + 40))
                self.end_headers()
                self.wfile.write(('Tampermonkey 服务不可用: %s' % e).encode('utf-8'))
            except Exception:
                pass

    def _send_short_query_notice(self, path, kw):
        # 2026-09-03: 关键词 ≤2 字符守卫 (gov_bigram 已下线, trigram FTS 需 ≥3 字符)
        _body = ('<!DOCTYPE html><html><head><meta charset="utf-8"><title>提示</title></head>'
                 '<body style="font-family:sans-serif;background:#f5f6fa;padding:60px 20px;text-align:center">'
                 '<div style="background:#fff;max-width:540px;margin:0 auto;padding:40px 32px;border-radius:10px;'
                 'box-shadow:0 2px 12px rgba(0,0,0,.08)">'
                 '<h2 style="margin-top:0;color:#333">关键词过短</h2>'
                 '<p style="color:#666;font-size:15px;line-height:1.8">关键词 “<b>' + html.escape(kw) + '</b>” 过短（≤2 个字符），'
                 '无法进行有效检索。请至少输入 <b>3 个字符</b>，建议使用完整的公司名称或更具体的项目关键词。</p>'
                 '<p><a href="' + html.escape(path) + '" style="color:#1a73e8;font-size:15px">← 返回搜索</a></p>'
                 '</div></body></html>')
        self.send_response(200)
        self.send_header('Content-Type', 'text/html; charset=utf-8')
        self.end_headers()
        self.wfile.write(_body.encode('utf-8'))

    def do_GET(self):

        parsed = urllib.parse.urlparse(self.path)

        params = urllib.parse.parse_qs(parsed.query)



        # Session check

        self.username = self._get_session_user()

        if parsed.path.startswith('/crawler/admin'):

            self.username = self.username or 'admin'  # bypass login for admin page

        # Tampermonkey script center proxy (/tm/* → 127.0.0.1:8010)

        if parsed.path.startswith('/tm/') or parsed.path == '/tm':

            self._proxy_tampermonkey(self.path)

            return

        if parsed.path in ('/login', '/logout'):

            if parsed.path == '/logout':

                self.send_response(302)

                self.send_header('Set-Cookie', 'session=; Path=/; Max-Age=0')

                self.send_header('Location', '/login')

                self.end_headers()

                return

            self.handle_login_page(params)

            return

        # Static files (no auth required)

        if parsed.path.startswith('/static/'):

            import os as _os, mimetypes

            _file_path = '/root/gov_crawler' + parsed.path

            if _os.path.isfile(_file_path):

                _content_type, _ = mimetypes.guess_type(_file_path)

                self.send_response(200)

                self.send_header('Content-Type', _content_type or 'application/octet-stream')

                self.send_header('Cache-Control', 'max-age=86400')

                self.end_headers()

                with open(_file_path, 'rb') as _f:

                    self.wfile.write(_f.read())

            else:

                self.send_response(404); self.end_headers(); self.wfile.write(b'404')

            return

        # Originals no-auth
        if parsed.path.startswith('/originals/detail'):
            self.handle_originals_detail(params)
            return
        if parsed.path.startswith('/originals'):
            self.handle_originals_page(params)
            return

        if not self.username:

            self.send_response(302)

            self.send_header('Location', '/login')

            self.end_headers()

            return





        self._log_usage(parsed.path, params)

        # 2026-09-03: 搜索词整体 ≤2 字符(去空格)守卫 — bigram 已恢复, 短词在组合内可由 bigram 命中; 仅整体过短不检索
        if parsed.path in ('/', '/crawler', '/crawler/'):
            _kws = [params.get(_k, [''])[0].strip() for _k in ('q', 'q2', 'q3')]
            _kws = [_k for _k in _kws if _k]
            if _kws and len(''.join(_k.replace(' ', '') for _k in _kws)) <= 2:
                self._send_short_query_notice(parsed.path, _kws[0])
                return

        # Track project visits on detail pages
        # ⚠️ /xxx/project/ 路径**不**在此跟踪——handle_db_detail 内部已 _track_project_visit（2506行），
        #    这里再记一次会导致每次点击记录 2 次。
        #    仅 /detail (handle_detail, 内部无跟踪) 在此记录。
        _parsed_path = parsed.path

        _params = params

        if _parsed_path == '/detail' and 'id' in _params:

            _id = _params.get('id', ['0'])[0]

            _track_project_visit('gov', _id, getattr(self, 'username', 'anonymous'))



        # ZNLH routes

        if parsed.path in ('/znlh', '/znlh/'):

            self.handle_db_list('znlh', params)

        elif parsed.path.startswith('/znlh/project/'):

            self.handle_db_detail('znlh', parsed.path)

        # ZC routes

        elif parsed.path in ('/zc', '/zc/'):

            # 2026-09-11: /zc 接入列表页 v2；?classic=1 回落旧版分页
            if params.get('classic'):
                self.handle_db_list('zc', params)
            else:
                self.send_response_html(200, render_list_shell(
                    username=getattr(self, 'username', '') or '',
                    daterange=(params.get('range') or ['all'])[0],
                    q=(params.get('q') or [''])[0],
                    route='zc'))

        elif parsed.path.startswith('/zc/project/'):

            self.handle_db_detail('zc', parsed.path)


        # ── 列表页 v2 API（2026-09-10）：游标分页 / 详情片段 / ⌘K 快速搜索 ──
        elif parsed.path == '/api/list':

            if not getattr(self, 'username', None):
                self.send_json({'ok': False, 'error': 'auth'}); return
            d = browse_route(
                (params.get('db') or ['crawler'])[0],
                cursor=(params.get('cursor') or [''])[0],
                limit=(params.get('limit') or ['30'])[0],
                direction=(params.get('dir') or ['next'])[0],
                daterange=(params.get('range') or ['all'])[0],
                industry=(params.get('industry') or [''])[0],
                q=(params.get('q') or [''])[0],
                kw=admin_whitelist_kw(),
                with_total=not (params.get('cursor') or [''])[0])
            d['ok'] = True
            self.send_json(d)

        elif parsed.path == '/api/item':

            if not getattr(self, 'username', None):
                self.send_json({'ok': False, 'error': 'auth'}); return
            iid = (params.get('id') or [''])[0]
            _db = (params.get('db') or ['crawler'])[0]
            if not iid:
                self.send_json({'ok': False, 'error': 'missing id'}); return
            _html = ''
            if _db == 'ccpc':
                # 中项网：抽屉片段由 render_ccpc_frag 单独产出（不动线上整页 handler）
                try:
                    _html = render_ccpc_frag(iid)
                except Exception:
                    _html = ''
            elif _db == 'contact':
                # 联系人：全字段卡片 + 关联项目反查
                try:
                    _html = render_contact_frag(iid)
                except Exception:
                    _html = ''
            else:
                # crawler / zc / znlh / eia：复用既有详情页的片段模式（?frag=1）
                _cap = {}
                _orig = self.send_response_html
                self.send_response_html = lambda code, html: _cap.update({'code': code, 'html': html})
                self._detail_frag = True
                try:
                    self.handle_db_detail(_db, '/%s/project/%s' % (_db, iid))
                except Exception:
                    _cap = {'code': 500, 'html': ''}
                finally:
                    self.send_response_html = _orig
                    self._detail_frag = False
                _html = _cap.get('html') or ''
            _meta = route_item_meta(_db, iid)
            self.send_json({'ok': bool(_html), 'html': _html, 'id': iid, **_meta})

        elif parsed.path == '/api/quicksearch':

            if not getattr(self, 'username', None):
                self.send_json({'ok': False, 'error': 'auth'}); return
            d = quicksearch_route((params.get('db') or ['crawler'])[0],
                                  (params.get('q') or [''])[0],
                                  (params.get('limit') or ['12'])[0])
            d['ok'] = True
            self.send_json(d)

        # Tools routes（2026-09-11 新增：实用工具市场；需登录）
        elif parsed.path in ('/tools', '/tools/'):
            self.send_response_html(200, render_tools_page(getattr(self, 'username', '') or ''))
            return

        elif parsed.path.startswith('/tools/download/'):
            _tid = urllib.parse.unquote(parsed.path[len('/tools/download/'):])
            _ok, _fpath, _err = tools_build_zip(_tid)
            if not _ok:
                _b = ('工具不存在或不可下载：' + (_err or '')).encode('utf-8')
                self.send_response(404)
                self.send_header('Content-Type', 'text/plain; charset=utf-8')
                self.send_header('Content-Length', str(len(_b)))
                self.end_headers()
                self.wfile.write(_b)
                return
            _fname = _os_tools.path.basename(_fpath)
            self.send_response(200)
            self.send_header('Content-Type', 'application/zip' if _fname.lower().endswith('.zip')
                             else 'application/octet-stream')
            self.send_header('Content-Length', str(_os_tools.path.getsize(_fpath)))
            self.send_header('Content-Disposition',
                             'attachment; filename="%s"; filename*=UTF-8\'\'%s'
                             % (_tools_ascii_name(_fname), urllib.parse.quote(_fname)))
            self.end_headers()
            with open(_fpath, 'rb') as _f:
                while True:
                    _chunk = _f.read(262144)
                    if not _chunk:
                        break
                    self.wfile.write(_chunk)
            return

        # Crawler routes

        elif parsed.path in ('/crawler', '/crawler/'):

            # 2026-09-10: 列表页 v2（左列表 + 右抽屉 + 无限滚动 + 游标分页）；?classic=1 回落旧版
            if params.get('classic'):
                self.handle_db_list('crawler', params)
            else:
                self.send_response_html(200, render_list_shell(
                    username=getattr(self, 'username', '') or '',
                    daterange=(params.get('range') or ['all'])[0],
                    industry=(params.get('industry') or [''])[0],
                    q=(params.get('q') or [''])[0]))

        elif parsed.path.startswith('/crawler/project/'):

            self.handle_db_detail('crawler', parsed.path)

        # EIA routes

        elif parsed.path in ('/eia', '/eia/'):

            self.handle_db_list('eia', params)

        elif parsed.path.startswith('/eia/project/'):

            self.handle_db_detail('eia', parsed.path)

        # Original routes

        elif parsed.path == '/':

            self.handle_search_page(params)

        elif parsed.path == '/detail':

            self.handle_detail(params)

        elif parsed.path == '/cceup':

            self.handle_cceup_page(params)

        elif parsed.path in ('/contact', '/contact/'):

            # 2026-09-11: /contact 接入列表页 v2（无日期列 → 单键 keyset）；
            # ?classic=1 回落旧版卡片分页
            if params.get('classic'):
                self.handle_contact_page(params)
            else:
                self.send_response_html(200, render_list_shell(
                    username=getattr(self, 'username', '') or '',
                    q=(params.get('q') or [''])[0],
                    route='contact'))

        elif parsed.path == '/ccpc':

            # 2026-09-11: /ccpc 接入列表页 v2；?classic=1 回落旧版分页
            if params.get('classic'):
                self.handle_cceup_projects(params)
            else:
                self.send_response_html(200, render_list_shell(
                    username=getattr(self, 'username', '') or '',
                    daterange=(params.get('range') or ['all'])[0],
                    q=(params.get('q') or [''])[0],
                    route='ccpc'))

        elif parsed.path == '/cceup/projects':

            self.handle_cceup_projects(params)

        # Temp DB route (crawler temporary storage, pre-sync)
        elif parsed.path in ('/crawler/temp', '/crawler/temp/'):
            self.handle_temp_db(params)
        elif parsed.path == '/crawler/temp/sync':
            self.handle_temp_sync()
        
        elif parsed.path.startswith('/cceup/project/') or parsed.path.startswith('/ccpc/project/'):

            self.handle_cceup_project_detail(self.path)

        elif parsed.path == '/dashboard':

            self.handle_dashboard()

        elif parsed.path == '/api/sources':

            self.handle_sources_api()

        elif parsed.path == '/api/visit':

            self.handle_visit(params)

        elif parsed.path == '/preview_md':

            self.send_preview_md()

        elif parsed.path == '/crawler/admin':

            # 2026-09-10: 脚本表服务端分页/筛选/排序（每页 10 行，避免 2.7MB HTML + 浏览器重排）
            try:
                _p = int((params.get('p') or ['1'])[0] or 1)
            except Exception:
                _p = 1
            self.send_response_html(200, render_admin_page(
                page=_p,
                fgroup=(params.get('g') or ['all'])[0],
                fstatus=(params.get('s') or ['all'])[0],
                fkw=(params.get('kw') or [''])[0],
                sort=(params.get('sort') or ['latest'])[0],
                sdir=(params.get('dir') or ['desc'])[0],
                username=getattr(self, 'username', '') or '',
            ))

        
        elif parsed.path == '/crawler/admin/run-daily':

            import json

            result = run_daily_crawl()

            self.send_response(200)

            self.send_header('Content-Type', 'application/json; charset=utf-8')

            self.end_headers()

            self.wfile.write(json.dumps(result, ensure_ascii=False).encode())

        elif parsed.path == '/crawler/admin/run-status':

            import json

            result = get_daily_run_status()

            self.send_response(200)

            self.send_header('Content-Type', 'application/json; charset=utf-8')

            self.end_headers()

            self.wfile.write(json.dumps(result, ensure_ascii=False).encode())

        elif parsed.path == '/crawler/admin/set-industry':

            import json

            n = params.get('name', [''])[0]
            ind = params.get('industry', [''])[0]
            result = set_crawler_industry(n, ind)

            self.send_response(200)

            self.send_header('Content-Type', 'application/json; charset=utf-8')
            self.send_header('Access-Control-Allow-Origin', '*')

            self.end_headers()

            self.wfile.write(json.dumps(result, ensure_ascii=False).encode())

        elif parsed.path == '/crawler/admin/toggle-enabled':
            import json
            n = params.get('name', [''])[0]
            result = toggle_crawler_enabled(n)
            self.send_response(200)
            self.send_header('Content-Type', 'application/json; charset=utf-8')
            self.send_header('Access-Control-Allow-Origin', '*')
            self.end_headers()
            self.wfile.write(json.dumps(result, ensure_ascii=False).encode())

        elif parsed.path == '/crawler/admin/save-exclude':
            import json, os
            keywords = params.get('kw', [''])[0]
            try:
                with open(ADMIN_EXCLUDE_PATH, 'w') as f:
                    f.write(keywords)
                result = {'success': True, 'keywords': keywords}
            except Exception as e:
                result = {'success': False, 'error': str(e)}
            self.send_response(200)
            self.send_header('Content-Type', 'application/json; charset=utf-8')
            self.end_headers()
            self.wfile.write(json.dumps(result, ensure_ascii=False).encode())

        elif parsed.path == "/crawler/admin/analysis":

            import json

            from crawler_admin import get_analysis_data

            period = params.get("period", ["day"])[0]

            mode = params.get("mode", ["incremental"])[0]

            metric = params.get("metric", ["both"])[0]

            days = int(params.get("days", ["90"])[0])

            result = get_analysis_data(period, mode, metric, days)

            self.send_response(200)

            self.send_header("Content-Type", "application/json; charset=utf-8")

            self.send_header("Access-Control-Allow-Origin", "*")

            self.end_headers()

            self.wfile.write(json.dumps(result, ensure_ascii=False).encode())

        else:
            self.send_response(404); self.end_headers(); self.wfile.write(b"404")



    def handle_originals_page(self, params):
        q = self.get_param(params, 'q', '').strip()
        page = int(self.get_param(params, 'page', '1'))
        page_size = int(self.get_param(params, 'page_size', '20'))
        offset = (page - 1) * page_size

        ORIGINALS_DB = '/root/originals.db'
        if not os.path.exists(ORIGINALS_DB):
            self.send_response(200)
            self.send_header('Content-type', 'text/html; charset=utf-8')
            self.end_headers()
            self.wfile.write(('<html><body><h2>Originals DB not found</h2></body></html>').encode('utf-8'))
            return

        conn = sqlite3.connect(ORIGINALS_DB)
        c = conn.cursor()

        results = []
        total = 0
        if q:
            inc_terms, exc_terms = parse_google_query(q)
            conds = []
            params = []
            for t in inc_terms:
                conds.append('(title LIKE ? OR content LIKE ? OR id_code LIKE ?)')
                params.extend(['%' + t + '%'] * 3)
            for t in exc_terms:
                conds.append('(title NOT LIKE ? AND content NOT LIKE ? AND id_code NOT LIKE ?)')
                params.extend(['%' + t + '%'] * 3)
            where = ' AND '.join(conds) if conds else '1=0'

            total = c.execute('SELECT COUNT(*) FROM tsk_data WHERE ' + where, params).fetchone()[0]
            rows = c.execute('SELECT id, title, id_code, source, content FROM tsk_data WHERE ' + where + ' ORDER BY id LIMIT ? OFFSET ?',
                params + [page_size, offset]).fetchall()

            for r in rows:
                results.append({'id': r[0], 'title': r[1], 'id_code': r[2] or '', 'source': r[3], 'content': r[4] or ''})

        # Get total count before closing
        total_count = 0
        try:
            total_count = c.execute('SELECT COUNT(*) FROM tsk_data').fetchone()[0]
        except:
            pass
        conn.close()

        self.send_response(200)
        self.send_header('Content-type', 'text/html; charset=utf-8')
        self.end_headers()

        h = '<html><head><meta charset=utf-8><title>Originals_Data</title>'
        h += '<style>body{font-family:sans-serif;margin:20px;background:#f5f5f5;text-align:center}.sb{margin:20px auto;display:flex;gap:0;justify-content:center;max-width:600px}.sb input{flex:1;padding:12px 20px;border:1px solid #dfe1e5;border-radius:24px 0 0 24px;font-size:16px;outline:none;max-width:500px}.sb input:focus{border-color:#1a73e8}.sb button{padding:12px 24px;background:#1a73e8;color:#fff;border:none;border-radius:0 24px 24px 0;font-size:15px;cursor:pointer}.sb button:hover{background:#1557b0}.results{max-width:800px;margin:0 auto;text-align:left}.result-item{padding:12px 0;border-bottom:1px solid #eee}.result-item h3{font-size:16px;font-weight:400;margin:0 0 2px 0}.result-item h3 a{color:#1a0dab;text-decoration:none}.result-item h3 a:hover{text-decoration:underline}.meta{font-size:12px;color:#70757a;display:flex;gap:12px;align-items:center}.id_code{color:#c00;font-weight:bold}.src{display:inline-block;padding:2px 8px;border-radius:3px;font-size:11px}.src-p{background:#e8f5e9;color:#2e7d32}.src-c{background:#e3f2fd;color:#1565c0}.src-l{background:#fff3e0;color:#e65100}.nav{margin:16px 0}.nav a,.nav strong{margin:0 3px;font-size:14px}.nav a{color:#1a73e8;text-decoration:none}.count{font-size:14px;color:#70757a;margin:8px 0;text-align:center}</style></head><body>'
        h += '<h1 style=text-align:center>🔍 原始数据库</h1>'
        h += '<p style=color:#666>' + str(total_count) + ' records &middot; Pharma / CPI / Plant</p>'
        h += '<form class=sb method=get action=/originals/>'
        h += '<input type=text name=q value="' + html.escape(q) + '" placeholder="输入关键词搜索...">'
        h += '<button type=submit>搜索</button></form>'

        if q:
            h += '<p class=count>' + str(total) + ' 条结果</p>'
            if results:
                h += '<div class=results>'
                for r in results:
                    sc = r['source']
                    sl = {'pharma':'P Pharma','cpi':'C CPI','plant':'L Plant'}.get(sc, sc)
                    sc_cls = {'pharma':'src-p','cpi':'src-c','plant':'src-l'}.get(sc, '')
                    h += '<div class=result-item>'
                    h += '<h3><a href="/originals/detail?id=' + str(r['id']) + '">' + html.escape(r['title']) + '</a></h3>'
                    h += '<div class=meta>'
                    h += '<span class=id_code>' + (html.escape(r['id_code']) if r['id_code'] else '-') + '</span>'
                    h += '<span class="src ' + sc_cls + '">' + sl + '</span>'
                    h += '</div></div>'
                h += '</div>'

                tp = max(1, (total + page_size - 1) // page_size)
                if tp > 1:
                    h += '<div class=nav>'
                    for p in range(1, min(tp + 1, 51)):
                        if p == page:
                            h += '<strong>' + str(p) + '</strong> '
                        else:
                            h += '<a href="/originals/?q=' + urllib.parse.quote(q) + '&page=' + str(p) + '">' + str(p) + '</a> '
                    if tp > 50:
                        h += '... <a href="/originals/?q=' + urllib.parse.quote(q) + '&page=' + str(tp) + '">Last</a>'
                    h += '</div>'
            else:
                h += '<p>无结果</p>'

        h += '<p style=color:#999;margin-top:30px><a href=/>"U+2190" 返回主页</a></p>'
        h += '</body></html>'
        self.wfile.write(h.encode('utf-8'))


    def handle_originals_detail(self, params):
        record_id = int(self.get_param(params, 'id', '0'))
        ORIGINALS_DB = '/root/originals.db'
        if not os.path.exists(ORIGINALS_DB):
            self.send_response_html(404, 'Not found')
            return
        conn = sqlite3.connect(ORIGINALS_DB)
        row = conn.execute('SELECT id, title, id_code, source, content FROM tsk_data WHERE id=?', (record_id,)).fetchone()
        conn.close()
        if not row:
            self.send_response_html(404, 'Not found')
            return
        r = {'id': row[0], 'title': row[1], 'id_code': row[2] or '', 'source': row[3], 'content': row[4] or ''}
        sc = r['source']
        sl = {'pharma':'P Pharma','cpi':'C CPI','plant':'L Plant'}.get(sc, sc)
        sc_cls = {'pharma':'src-p','cpi':'src-c','plant':'src-l'}.get(sc, '-')

        h = '<html><head><meta charset=utf-8><title>' + html.escape(r['title'][:80]) + '</title>'
        h += '<style>body{font-family:sans-serif;margin:20px;background:#f5f5f5}a{color:#1a73e8}.back{font-size:13px;color:#70757a;margin-bottom:12px;display:block}.card{background:#fff;border-radius:8px;padding:24px;margin-bottom:16px;box-shadow:0 1px 3px rgba(0,0,0,.08);max-width:780px;margin-left:auto;margin-right:auto}.card h1{font-size:22px;margin-bottom:8px}.meta{font-size:13px;color:#70757a;display:flex;gap:12px;margin-bottom:16px}.id_code{color:#c00;font-weight:bold}.src{display:inline-block;padding:2px 8px;border-radius:3px;font-size:11px}.src-p{background:#e8f5e9;color:#2e7d32}.src-c{background:#e3f2fd;color:#1565c0}.src-l{background:#fff3e0;color:#e65100}.content{font-size:14px;line-height:1.8;color:#333;white-space:pre-wrap;word-break:break-word;overflow-x:auto}</style></head><body>'
        h += '<a class=back href="/originals/">&larr; 返回列表</a>'
        h += '<div class=card>'
        h += '<h1>' + html.escape(r['title']) + '</h1>'
        h += '<div class=meta>'
        h += '<span class=id_code>' + (html.escape(r['id_code']) if r['id_code'] else '-') + '</span>'
        h += '<span class="src ' + sc_cls + '">' + sl + '</span>'
        h += '<span>ID: ' + str(r['id']) + '</span>'
        h += '</div>'
        h += '<div class=content>' + html.escape(re.sub(r'\n{3,}', '\n\n', r['content'] or '')) + '</div>'
        h += '</div></body></html>'
        self.send_response_html(200, h)

    def handle_temp_db(self, params):
        """Show temp_search.db contents with search."""
        TEMP_DB = '/root/temp_search.db'
        TRACKING_FILE = '/root/.temp_sync_state.json'

        page = int(self.get_param(params, 'page', '1'))
        per = 20
        offset = (page - 1) * per
        q = self.get_param(params, 'q', '').strip()

        sync_state = {}
        if os.path.exists(TRACKING_FILE):
            try:
                with open(TRACKING_FILE) as f:
                    sync_state = json.load(f)
            except:
                pass

        rows = []
        total = 0
        temp_size = 0
        last_synced = sync_state.get('last_synced_id', 0)
        error = None
        un_synced = 0

        if os.path.exists(TEMP_DB):
            temp_size = os.path.getsize(TEMP_DB)
            try:
                db = sqlite3.connect(TEMP_DB)
                db.execute('PRAGMA busy_timeout=3000')
                c = db.cursor()

                if q:
                    like = '%' + q + '%'
                    total = c.execute(
                        'SELECT COUNT(*) FROM gov_raw WHERE title LIKE ? OR site_name LIKE ? OR page_url LIKE ?',
                        (like, like, like)
                    ).fetchone()[0]
                    un_synced = c.execute(
                        'SELECT COUNT(*) FROM gov_raw WHERE id > ? AND (title LIKE ? OR site_name LIKE ? OR page_url LIKE ?)',
                        (last_synced, like, like, like)
                    ).fetchone()[0]
                    c.execute(
                        'SELECT id, site_name, title, publish_date FROM gov_raw WHERE title LIKE ? OR site_name LIKE ? OR page_url LIKE ? ORDER BY id DESC LIMIT ? OFFSET ?',
                        (like, like, like, per, offset)
                    )
                else:
                    total = c.execute('SELECT COUNT(*) FROM gov_raw').fetchone()[0]
                    un_synced = c.execute('SELECT COUNT(*) FROM gov_raw WHERE id > ?', (last_synced,)).fetchone()[0]
                    c.execute(
                        'SELECT id, site_name, title, publish_date FROM gov_raw ORDER BY id DESC LIMIT ? OFFSET ?',
                        (per, offset)
                    )

                rows = [{'id': r[0], 'site': r[1], 'title': r[2], 'date': r[3]} for r in c.fetchall()]
                db.close()
            except Exception as e:
                error = str(e)

        # Build HTML (string concat to avoid quoting issues)
        h = '<!DOCTYPE html><html lang="zh-CN"><head>'
        h += '<meta charset="UTF-8"><meta name="viewport" content="width=device-width,initial-scale=1">'
        h += '<title>Temp DB - 临时数据</title><style>'
        h += '*{margin:0;padding:0;box-sizing:border-box}'
        h += 'body{font-family:-apple-system,\"Microsoft YaHei\",sans-serif;background:#f0f2f5;color:#333}'
        h += '.container{max-width:1000px;margin:0 auto;padding:20px}'
        h += '.header{display:flex;justify-content:space-between;align-items:center;margin-bottom:16px}'
        h += '.header h1{font-size:20px;color:#1a1a2e}'
        h += '.stats{display:flex;gap:12px;margin-bottom:16px;flex-wrap:wrap}'
        h += '.stat{padding:12px 18px;border-radius:10px;font-size:13px;min-width:120px}'
        h += '.stat.blue{background:#e3f0ff;color:#1967d2}'
        h += '.stat.green{background:#e6f4ea;color:#137333}'
        h += '.stat.orange{background:#fef7e0;color:#e37400}'
        h += '.stat .num{font-size:20px;font-weight:700}'
        h += '.search-bar{display:flex;gap:8px;margin-bottom:16px}'
        h += '.search-bar input{flex:1;padding:10px 14px;border:1px solid #dadce0;border-radius:8px;font-size:14px;outline:0}'
        h += '.search-bar input:focus{border-color:#1967d2}'
        h += '.search-bar button{padding:10px 20px;background:#1967d2;color:#fff;border:none;border-radius:8px;cursor:pointer;font-size:14px}'
        h += 'table{width:100%;border-collapse:collapse;background:#fff;border-radius:10px;overflow:hidden}'
        h += 'th,td{padding:10px 14px;text-align:left;font-size:13px;border-bottom:1px solid #eee}'
        h += 'th{background:#f8f9fa;font-weight:600;color:#555}'
        h += 'tr:hover{background:#f8f9ff}'
        h += 'td.sn{max-width:160px;overflow:hidden;text-overflow:ellipsis;white-space:nowrap}'
        h += 'td.tl{max-width:400px;overflow:hidden;text-overflow:ellipsis;white-space:nowrap}'
        h += '.paging{display:flex;gap:8px;justify-content:center;flex-wrap:wrap;margin-top:16px}'
        h += '.paging a{padding:6px 14px;border:1px solid #dadce0;border-radius:6px;text-decoration:none;color:#1967d2;font-size:13px}'
        h += '.paging a:hover{background:#e8f0fe}'
        h += '.paging .cur{padding:6px 14px;border:1px solid #1967d2;border-radius:6px;background:#1967d2;color:#fff;font-size:13px}'
        h += '.btn{padding:8px 18px;border:none;border-radius:6px;cursor:pointer;font-size:13px;text-decoration:none}'
        h += '.btn.primary{background:#1967d2;color:#fff}'
        h += '.btn.outline{border:1px solid #dadce0;background:#fff;color:#333}'
        h += 'code{background:#f1f3f4;padding:2px 6px;border-radius:4px;font-size:12px}'
        h += '.hint{color:#888;font-size:12px;margin-top:4px}'
        h += '</style></head><body>'
        h += '<div class=container>'
        h += '<div class=header><h1>🧊 Temp DB</h1>'
        h += '<a href=/crawler/ class=btn.outline>← 主库</a></div>'

        if error:
            h += '<div class=stat.red>' + str(error) + '</div>'
        else:
            h += '<div class=stats>'
            label = '结果' if q else '总行数'
            h += '<div class=stat.blue><div class=num>' + str(total) + '</div>' + label + '</div>'
            h += '<div class=stat.orange><div class=num>' + str(un_synced) + '</div>待同步</div>'
            last_sv = sync_state.get('last_sync') or '—'
            h += '<div class=stat.green><div class=num>' + last_sv + '</div>上次同步</div>'
            h += '<div class=stat><div class=num>' + str(temp_size // 1024) + 'KB</div>文件大小</div>'
            h += '</div>'

            # Search bar
            qs = q.replace('"', '&quot;')
            h += '<form class=search-bar method=get action=/crawler/temp/>'
            h += '<input type=text name=q placeholder=搜索标题/站点/链接... value=\"' + qs + '\">'
            h += '<button type=submit>🔍 搜索</button></form>'

            # Action buttons
            h += '<div style=margin-bottom:12px;display:flex;gap:8px>'
            h += '<a href=/crawler/temp/sync class=btn.primary>🔄 立即同步</a>'
            clear_label = '🔄 刷新' if not q else '✖ 清除搜索'
            h += '<a href=/crawler/temp/ class=btn.outline>' + clear_label + '</a>'
            h += '</div>'

            if q:
                h += '<div class=hint>搜索 "' + q + '", 共 ' + str(total) + ' 条结果</div>'

            if rows:
                h += '<table><thead><tr><th>ID</th><th>站点</th><th>标题</th><th>日期</th></tr></thead><tbody>'
                for r in rows:
                    h += '<tr><td>' + str(r['id']) + '</td>'
                    site = (r['site'] or '')[:80]
                    title = (r['title'] or '')[:200]
                    h += '<td class=sn>' + site + '</td>'
                    h += '<td class=tl>' + title + '</td>'
                    h += '<td>' + (r['date'] or '') + '</td></tr>'
                h += '</tbody></table>'

                total_pages = max(1, (total + per - 1) // per)
                if total_pages > 1:
                    h += '<div class=paging>'
                    for p in range(1, total_pages + 1):
                        link = '/crawler/temp/?page=' + str(p)
                        if q:
                            link += '&q=' + q
                        if p == page:
                            h += '<span class=cur>' + str(p) + '</span>'
                        else:
                            h += '<a href=\"' + link + '\">' + str(p) + '</a>'
                    h += '</div>'
            else:
                if q:
                    h += '<div class=stat.green>未找到匹配结果</div>'
                else:
                    h += '<div class=stat.green>暂无数据</div>'

        h += '<div style=margin-top:20px;font-size:12px;color:#888>'
        h += '<code>temp_search.db</code> | last_synced_id=' + str(last_synced)
        h += ' | 每日8:00自动同步</div>'
        h += '</div></body></html>'

        self.send_response_html(200, h)



    def handle_temp_sync(self):
        """Trigger immediate sync from temp to main."""
        import subprocess
        try:
            r = subprocess.run(['python3', '/root/sync_temp_to_main.py'], capture_output=True, text=True, timeout=30)
            output = r.stdout or ''
            err = r.stderr or ''
            if err:
                output += f'\nSTDERR: {err}'
            if r.returncode != 0:
                output += f'\nExit code: {r.returncode}'
        except Exception as e:
            output = f'Error: {e}'
        
        self.send_response(200)
        self.send_header('Content-Type', 'text/plain; charset=utf-8')
        self.end_headers()
        self.wfile.write(output.encode())

    def do_POST(self):

        length = int(self.headers.get('Content-Length', 0))

        body = self.rfile.read(length).decode('utf-8', errors='replace') if length else ''

        params = urllib.parse.parse_qs(body)

        path = urllib.parse.urlparse(self.path).path

        if path == '/login':

            username = (params.get('username') or [''])[0].strip()

            if username in ALLOWED_USERS:

                token = str(uuid.uuid4())

                db = sqlite3.connect(SESSION_DB)

                db.execute('INSERT OR REPLACE INTO sessions (token, username) VALUES (?,?)', (token, username))

                db.commit(); db.close()

                self.send_response(302)

                self.send_header('Set-Cookie', f'session={token}; Path=/; Max-Age=2592000')

                self.send_header('Location', '/')

                self.end_headers()

            else:

                self.send_response(302)

                self.send_header('Location', '/login?error=1')

                self.end_headers()

        elif path == '/crawler/admin/run-daily':

            import json

            result = run_daily_crawl()

            self.send_response(200)

            self.send_header('Content-Type', 'application/json; charset=utf-8')

            self.end_headers()

            self.wfile.write(json.dumps(result, ensure_ascii=False).encode())

        elif path == '/crawler/admin/run-status':

            import json

            result = get_daily_run_status()

            self.send_response(200)

            self.send_header('Content-Type', 'application/json; charset=utf-8')

            self.end_headers()

            self.wfile.write(json.dumps(result, ensure_ascii=False).encode())

        else:

            self.send_response(404); self.end_headers(); self.wfile.write(b'404')



    def get_param(self, params, name, default=''):

        vals = params.get(name, [default])

        return vals[0] if vals else default



    def _get_session_user(self):

        cookie = self.headers.get('Cookie', '')

        for part in cookie.split(';'):

            part = part.strip()

            if part.startswith('session='):

                token = part[8:]

                db = sqlite3.connect(SESSION_DB)

                try:

                    row = db.execute('SELECT username FROM sessions WHERE token=?', (token,)).fetchone()

                    if row:

                        db.execute('UPDATE sessions SET last_seen=datetime("now","localtime") WHERE token=?', (token,))

                        db.commit()

                    return row[0] if row else None

                finally:

                    db.close()

        return None



    def _log_usage(self, path, params):

        if not self.username: return

        q = self.get_param(params, 'q', '')

        action = 'view'

        if q: action = 'search'

        elif path.startswith('/cceup/project/') or path.startswith('/crawler/project/') or path.startswith('/zc/project/') or path.startswith('/znlh/project/'):

            action = 'detail'

        db = sqlite3.connect(SESSION_DB)

        try:

            db.execute('INSERT INTO usage_log (username, action, page, query, ip) VALUES (?,?,?,?,?)',

                       (self.username, action, path, q[:200], self.client_address[0]))

            db.commit()

        except: pass

        finally: db.close()



    def handle_login_page(self, params):

        error = 'error' in params

        h = '<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>登录 - 项目信息检索</title>'

        h += '<style>*{margin:0;padding:0;box-sizing:border-box}body{font-family:-apple-system,"Microsoft YaHei",sans-serif;background:#f0f2f5;display:flex;justify-content:center;align-items:center;min-height:100vh}.card{background:#fff;border-radius:16px;padding:40px;width:380px;box-shadow:0 4px 24px rgba(0,0,0,.08);text-align:center}h1{font-size:22px;color:#1a1a2e;margin-bottom:8px}.sub{color:#888;font-size:13px;margin-bottom:28px}.inp{width:100%;padding:12px 16px;border:1px solid #ddd;border-radius:10px;font-size:15px;outline:none;transition:border .2s;box-sizing:border-box}.inp:focus{border-color:#4361ee}.btn{width:100%;padding:12px;background:#4361ee;color:#fff;border:none;border-radius:10px;font-size:15px;cursor:pointer;margin-top:16px}.btn:hover{background:#3651d4}.warn{color:#e53935;font-size:13px;margin-top:16px;padding:10px;background:#fff0f0;border-radius:8px}.tip{color:#999;font-size:12px;margin-top:18px}</style></head><body>'

        h += '<div class="card"><h1>🔍 项目信息检索</h1><p class="sub">请输入用户名登录</p>'

        if error: h += '<div class="warn">用户名无效，请重新输入</div>'

        h += '<form method="post" action="/login"><input class="inp" name="username" placeholder="输入用户名" autofocus>'

        h += '<button class="btn" type="submit">登录</button></form>'

        h += '<div class="tip">数据无价 请勿外传</div></div></body></html>'

        self.send_response_html(200, h)



    def send_preview_md(self):

        try:

            with open('/root/preview_md.html', 'rb') as f:

                data = f.read()

            self.send_response(200)

            self.send_header('Content-Type','text/html; charset=utf-8')

            self.send_header('Content-Length',str(len(data)))

            self.end_headers()

            self.wfile.write(data)

        except Exception as e:

            self.send_response(500)

            self.end_headers()

            self.wfile.write(str(e).encode())



    def send_response_html(self, code, html):

        b = html.encode()

        self.send_response(code)

        self.send_header('Content-Type','text/html; charset=utf-8')

        self.send_header('Content-Length',str(len(b)))

        self.end_headers()

        self.wfile.write(b)



    # ─── 通用数据库列表/详情 ───



    def handle_db_list(self, db_type, params):

        page = int(self.get_param(params, 'page', '1'))

        q = self.get_param(params, 'q', '').strip()

        q2 = self.get_param(params, 'q2', '').strip()

        q3 = self.get_param(params, 'q3', '').strip()

        daterange = self.get_param(params, 'range', 'all')
        industry_filter = self.get_param(params, 'industry', '')


        per = 20

        names = {'znlh':'中能联合','zc':'中策大数据','crawler':'EIA','eia':'EIA项目库'}

        title = names.get(db_type, db_type)

        total = 0

        if db_type == 'crawler':

            has_more, rows = False, []

            if os.path.exists(DB_PATH):

                db = get_db()

                c = db.cursor()

                fetch_size = per + 1

                offset = (page-1)*per

                # 构建白名单关键词 SQL 条件（数据库层面过滤，确保 total 正确）
                # 语义：不再排除，而是"仅限给定关键词"（OR 匹配任一即保留）
                _include_kw_list = []
                _include_cond_r = ''        # 带 r. 前缀（FTS/LIKE JOIN 路径用）
                _include_cond_plain = ''    # 无前缀（无词浏览路径用）
                _include_params = []
                try:
                    with open(ADMIN_EXCLUDE_PATH) as _ef:
                        _include_kw = _ef.read().strip()
                    if _include_kw:
                        # 逗号或空格分隔（兼容中英文逗号）
                        _include_list = [k.strip().lower() for k in re.split(r'[,，\s]+', _include_kw) if k.strip()]
                        if _include_list:
                            _include_kw_list = _include_list
                            # 与 v2 列表共用同一谓词构造（2026-09-11 口径统一）
                            _include_cond_r, _include_params = _whitelist_like_pred(_include_list, 'r.')
                            _include_cond_plain, _ = _whitelist_like_pred(_include_list, '')
                except:
                    pass


                # ── 2026-09-03 gov_entity 精确反查: 完整公司名/电话 (正文出现也命中) ──
                _ent_done = False
                if not q2 and not q3 and ' ' not in (q or '').strip():
                    _ep = None
                    if q and not q.startswith('-'):
                        _inc0, _exc0 = parse_google_query(q)
                        if len(_inc0) == 1 and not _exc0:
                            _t0 = _inc0[0]
                            if contains_chinese(_t0) and len(_t0) >= 3:
                                _ep = ('company', _t0)
                            else:
                                _dg = re.sub(r'\D', '', _t0)
                                if 7 <= len(_dg) <= 13:
                                    _ep = ('phone', _dg)
                    if _ep:
                        try:
                            _ec = ['e.etype = ?', 'e.value = ?']
                            _ev = [_ep[0], _ep[1]]
                            if daterange and daterange != 'all':
                                _days = {'1d': 1, '7d': 7, '30d': 30}[daterange]
                                _ec.append("(r.publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' AND SUBSTR(r.publish_date,1,10) >= date('now', '-%d days'))" % _days)
                            _if1, _ip1 = _ind_in_clause('r.industry', industry_filter)
                            if _if1:
                                _ec.append(_if1)
                                _ev.extend(_ip1)
                            if _include_kw_list:
                                _ec.append('(' + ' OR '.join('(r.title LIKE ? OR r.site_name LIKE ? OR r.page_url LIKE ?)' for _ in _include_kw_list) + ')')
                                _ev.extend([f'%{_k}%' for _k in _include_kw_list for _ in range(3)])
                            c.execute('SELECT COUNT(*) FROM gov_entity e JOIN gov_raw r ON r.id = e.doc_id WHERE ' + ' AND '.join(_ec), _ev)
                            total = c.fetchone()[0]
                            if total > 0:
                                c.execute('SELECT r.* FROM gov_entity e JOIN gov_raw r ON r.id = e.doc_id WHERE ' + ' AND '.join(_ec) + ' ORDER BY r.publish_date DESC, r.id DESC LIMIT ? OFFSET ?', _ev + [fetch_size, offset])
                                _ent_rows = [dict(r) for r in c.fetchall()]
                                rows = _ent_rows
                                _ent_done = True
                        except sqlite3.OperationalError:
                            pass
                if q and not _ent_done:

                    inc, exc = parse_google_query(q)

                    fts_parts = []

                    for t in inc:

                        fts_parts.append(t)

                    for t in exc:

                        fts_parts.append('-' + t)

                    fts_q = ' AND '.join(fts_parts)

                    extra_parts = []

                    extra_params = []

                    if q2:

                        extra_parts.append('(r.title LIKE ? OR r.source_url LIKE ? OR r.page_url LIKE ?)')

                        extra_params.extend([f'%{q2}%']*3)

                    if q3:

                        extra_parts.append('(r.title LIKE ? OR r.source_url LIKE ? OR r.page_url LIKE ?)')

                        extra_params.extend([f'%{q3}%']*3)

                    if daterange and daterange != 'all':

                        days = {'1d':1,'7d':7,'30d':30}[daterange]

                        extra_parts.append(f"(r.publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' AND SUBSTR(r.publish_date,1,10) >= date('now', '-{days} days'))")

                    _if2, _ip2 = _ind_in_clause('r.industry', industry_filter)
                    if _if2:
                        extra_parts.append('(' + _if2 + ')')
                        extra_params.extend(_ip2)
                    # 白名单关键词**不**加入 extra_parts —— bigram 快路径用集合差、FTS/LIKE 路径单独拼接。
                    # (避免 bigram COUNT/SELECT 里带白名单 LIKE %kw% 全表扫卡死)
                    extra_sql = ' AND ' + ' AND '.join(extra_parts) if extra_parts else ''
                    _include_sql_r = (' AND ' + _include_cond_r) if _include_cond_r else ''
                    _include_all_params = list(_include_params)

                    # ALL queries try FTS first (Chinese included)
                    # 中文单词>=3字符：先用 trigram FTS（比 LIKE 快100倍）
                    # 日期/行业过滤(extra_parts)直接拼进 FTS MATCH——不要 LIMIT 5000 截断(会丢新入库的高rowid记录)
                    if contains_chinese(q):
                        _fts_ok = False
                        if not exc and all(len(t) >= 3 for t in inc) and len(q.strip()) >= 3:
                            try:
                                _fts_q = sanitize_fts_query(q)
                                c.execute(f"SELECT COUNT(*) FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE gov_search MATCH ?{extra_sql}{_include_sql_r}", (_fts_q, *extra_params, *_include_all_params))
                                total = c.fetchone()[0]
                                c.execute(f"SELECT r.* FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE gov_search MATCH ?{extra_sql}{_include_sql_r} ORDER BY r.publish_date DESC, r.id DESC LIMIT ? OFFSET ?", (_fts_q, *extra_params, *_include_all_params, fetch_size, offset))
                                _fts_ok = True
                            except:
                                pass

                        if not _fts_ok:
                            # 先用FTS缩小范围，再在内存中过滤
                            try:
                                _fts_rowids = set()
                                for _t in inc:
                                    if len(_t) >= 3:
                                        c.execute("SELECT rowid FROM gov_search WHERE gov_search MATCH ? LIMIT 5000", (sanitize_fts_query(_t),))
                                        _ids = set(r[0] for r in c.fetchall())
                                        if _fts_rowids:
                                            _fts_rowids = _fts_rowids & _ids
                                        else:
                                            _fts_rowids = _ids
                                if _fts_rowids:
                                    _id_list = ",".join(str(i) for i in list(_fts_rowids)[:5000])
                                    _where_extra = " AND r.id IN (" + _id_list + ")"
                                else:
                                    _where_extra = ""
                                # exclusion handled via FTS NOT terms in query
                            except:
                                _where_extra = ""
                            # LIKE 条件（在FTS缩小后的ID范围内）
                            like_conds = []
                            like_params = []
                            for t in inc:
                                like_conds.append('(r.title LIKE ? OR r.site_name LIKE ? OR r.summary LIKE ?)')
                                like_params.extend(['%' + t + '%', '%' + t + '%', '%' + t + '%'])
                            for t in exc:
                                like_conds.append('(r.title NOT LIKE ? AND r.site_name NOT LIKE ? AND r.summary NOT LIKE ?)')
                                like_params.extend(['%' + t + '%', '%' + t + '%', '%' + t + '%'])
                            like_where = ' AND '.join(like_conds)
                            if extra_parts:
                                like_where += ' AND ' + ' AND '.join(extra_parts)
                                like_params += extra_params
                            if _include_cond_r:
                                like_where += _include_sql_r
                                like_params += _include_all_params
                            if _where_extra:
                                like_where += _where_extra
                            # bigram 快路径: inc 全是中文≥2字且有有效 gram 时, 用 gov_bigram 交集缩小范围
                            # 避免 LIKE %xx% 全表扫 + 白名单 LIKE 全表扫双重慢(45s+卡死)
                            # (2字词"环境"不满足 FTS trigram, 会走到这里)
                            _bigram_ok = False
                            # 2026-09-03: gov_bigram 已恢复 → 该快路径重新启用
                            if (inc and not _where_extra
                                    and all(len(t) >= 2 and contains_chinese(t) and token_grams(t) for t in inc)):
                                try:
                                    _b_sets = []
                                    for _t in inc:
                                        _g_s = None
                                        for _g in token_grams(_t):
                                            c.execute('SELECT rowid FROM gov_bigram WHERE gram = ?', (_g,))
                                            _s = set(r[0] for r in c.fetchall())
                                            _g_s = _s if _g_s is None else (_g_s & _s)
                                            if not _g_s:
                                                break
                                        _b_sets.append(_g_s or set())
                                    _b_ids = set.intersection(*_b_sets) if _b_sets else set()
                                    if _b_ids:
                                        # 白名单过滤 (bigram 只索引 title/site_name; page_url 白名单差异极少见)
                                        # OR 语义: 任一关键词命中即保留 → 各关键词集合取并集再交集
                                        if _include_kw_list:
                                            _kw_union = set()
                                            for _kw in _include_kw_list:
                                                _kw_grams = token_grams(_kw) or ([_kw] if len(_kw) == 2 else [])
                                                _kw_s = None
                                                for _g in _kw_grams:
                                                    c.execute('SELECT rowid FROM gov_bigram WHERE gram = ?', (_g,))
                                                    _s = set(r[0] for r in c.fetchall())
                                                    _kw_s = _s if _kw_s is None else (_kw_s & _s)
                                                if _kw_s is None:
                                                    _kw_s = set()
                                                _kw_union |= _kw_s
                                            _b_ids = _b_ids & _kw_union
                                        # 排除词 → bigram 集合差
                                        for _t in exc:
                                            for _g in token_grams(_t):
                                                c.execute('SELECT rowid FROM gov_bigram WHERE gram = ?', (_g,))
                                                _exc_s = set(r[0] for r in c.fetchall())
                                                _b_ids = _b_ids - _exc_s
                                        if _b_ids:
                                            # 大集合时禁止 IN 超参数; 用 EXISTS + 过滤条件流式计数
                                            # inc 的 gram 条件
                                            _b_gram_counts = [len(token_grams(t)) for t in inc]
                                            _b_ex_conds = ' AND '.join(
                                                ' AND '.join(f'EXISTS (SELECT 1 FROM gov_bigram b{i}_{j} WHERE b{i}_{j}.rowid=r.id AND b{i}_{j}.gram=?)'
                                                             for j in range(_b_gram_counts[i]))
                                                for i in range(len(inc)))
                                            _b_ex_params = []
                                            for _t in inc:
                                                _b_ex_params.extend(token_grams(_t))
                                            # 白名单 → OR of (AND of EXISTS gram per kw) —— 任一关键词命中即保留
                                            _b_wl_conds = []
                                            for _ki, _kw in enumerate(_include_kw_list):
                                                _b_kw_grams = token_grams(_kw) or ([_kw] if len(_kw) == 2 else [])
                                                if _b_kw_grams:
                                                    _b_wl_conds.append('(' + ' AND '.join(
                                                        f'EXISTS (SELECT 1 FROM gov_bigram w{_ki}_{j} WHERE w{_ki}_{j}.rowid=r.id AND w{_ki}_{j}.gram=?)'
                                                        for j in range(len(_b_kw_grams))) + ')')
                                                    _b_ex_params.extend(_b_kw_grams)
                                            # 排除词 → NOT EXISTS (含该词则排除)
                                            for _t in exc:
                                                for _g in (token_grams(_t) or ([_t] if len(_t) == 2 else [])):
                                                    _b_wl_conds.append('NOT EXISTS (SELECT 1 FROM gov_bigram e0 WHERE e0.rowid=r.id AND e0.gram=?)')
                                                    _b_ex_params.append(_g)
                                            if _b_wl_conds:
                                                _b_wl_sql = ' AND (' + ' OR '.join(_b_wl_conds) + ')'
                                            else:
                                                _b_wl_sql = ''
                                            # extra_parts (日期/行业) 直接拼入
                                            if extra_parts:
                                                _b_extra_join = ' AND ' + ' AND '.join(extra_parts)
                                            else:
                                                _b_extra_join = ''
                                            _b_count_sql = f'SELECT COUNT(*) FROM gov_raw r WHERE {_b_ex_conds}{_b_wl_sql}{_b_extra_join}'
                                            c.execute(_b_count_sql, _b_ex_params + extra_params if extra_parts else _b_ex_params)
                                            total = c.fetchone()[0]
                                            if total > 0:
                                                _b_limit_sql = f'SELECT r.id FROM gov_raw r WHERE {_b_ex_conds}{_b_wl_sql}{_b_extra_join} ORDER BY r.publish_date DESC, r.id DESC LIMIT ? OFFSET ?'
                                                c.execute(_b_limit_sql, (_b_ex_params + extra_params if extra_parts else _b_ex_params) + [fetch_size, offset])
                                                _b_page_ids = [r[0] for r in c.fetchall()]
                                                has_more = len(_b_page_ids) > per
                                                if has_more:
                                                    _b_page_ids = _b_page_ids[:per]
                                                if _b_page_ids:
                                                    _b_ph = ','.join('?' * len(_b_page_ids))
                                                    c.execute(f'SELECT r.* FROM gov_raw r WHERE r.id IN ({_b_ph})', _b_page_ids)
                                                    rows = [dict(r) for r in c.fetchall()]
                                                else:
                                                    rows = []
                                                _bigram_ok = True
                                except Exception:
                                    _bigram_ok = False
                            if not _bigram_ok:
                                c.execute(f'SELECT COUNT(*) FROM gov_raw r WHERE {like_where}', like_params)
                                total = c.fetchone()[0]
                                c.execute(f'SELECT r.*, "" as rank FROM gov_raw r WHERE {like_where} ORDER BY r.publish_date DESC, r.id DESC LIMIT ? OFFSET ?', like_params + [fetch_size, offset])

                    else:
                        try:

                            _fts_q = sanitize_fts_query(fts_q)
                            c.execute(f'SELECT COUNT(*) FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE gov_search MATCH ?{extra_sql}', (_fts_q, *extra_params))
                            total = c.fetchone()[0]
                            c.execute(f'SELECT r.* FROM gov_search s JOIN gov_raw r ON s.rowid=r.id WHERE gov_search MATCH ?{extra_sql} ORDER BY r.publish_date DESC, r.id DESC LIMIT ? OFFSET ?', (_fts_q, *extra_params, fetch_size, offset))

                        except sqlite3.OperationalError:

                            # FTS语法错误，降级到LIKE

                            like_q = '%' + q + '%'

                            c.execute('SELECT COUNT(*) FROM gov_raw r WHERE (r.title LIKE ? OR r.summary LIKE ?) ORDER BY r.publish_date DESC, r.id DESC LIMIT ? OFFSET ?', (like_q, like_q, fetch_size, offset))
                            total = 0
                            rows = [dict(r) for r in c.fetchall()]
                            # 无法直接用FTS加速的fallback，白名单过滤（仅限给定关键词）
                            if _include_kw_list:
                                rows = [r for r in rows if any(kw in (r.get('title','') or '').lower() or kw in (r.get('site_name','') or '').lower() or kw in (r.get('page_url','') or '').lower() for kw in _include_kw_list)]

                elif not q:

                    # 无搜索词：按日期浏览

                    where_parts = []

                    sql_params = []

                    if q2:

                        where_parts.append('(title LIKE ? OR source_url LIKE ? OR page_url LIKE ?)')

                        sql_params.extend([f'%{q2}%']*3)

                    if q3:

                        where_parts.append('(title LIKE ? OR source_url LIKE ? OR page_url LIKE ?)')

                        sql_params.extend([f'%{q3}%']*3)

                    if daterange and daterange != 'all':

                        days = {'1d':1,'7d':7,'30d':30}[daterange]

                        where_parts.append(f"(publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' AND SUBSTR(publish_date,1,10) >= date('now', '-{days} days'))")

                    _if3, _ip3 = _ind_in_clause('industry', industry_filter)
                    if _if3:
                        where_parts.append(_if3)
                        sql_params.extend(_ip3)
                    # 白名单关键词（无词浏览路径，无 r. 前缀）
                    if _include_cond_plain:
                        where_parts.append(_include_cond_plain)
                        sql_params.extend(_include_params)
                    where_sql = ' WHERE ' + ' AND '.join(where_parts) if where_parts else ''

                    # 无词浏览 + 白名单: LIKE %kw% 无法用索引, 全表 COUNT 54s 卡死。
                    # 改用截断式计数: ORDER BY publish_date 走日期索引流式, 数到 5001 就停 (0.00s)。
                    # 超过 5000 条时 total 显示 5001(截断), admin 页展示"共 5001+ 条"语义。
                    if _include_kw_list:
                        c.execute(f'SELECT COUNT(*) FROM (SELECT id FROM gov_raw{where_sql} ORDER BY publish_date DESC, id DESC LIMIT 5001)', sql_params)
                        total = c.fetchone()[0]
                    else:
                        c.execute(f'SELECT COUNT(*) FROM gov_raw{where_sql}', sql_params)
                        total = c.fetchone()[0]
                    c.execute(f'SELECT * FROM gov_raw{where_sql} ORDER BY publish_date DESC, id DESC LIMIT ? OFFSET ?', sql_params + [fetch_size, offset])

                rows = [dict(r) for r in c.fetchall()]
                # Post-fetch whitelist filter (applies to both query and no-query paths)
                # 白名单语义：仅保留匹配任一给定关键词的行（title/site_name/page_url）
                if _include_kw_list:
                    rows = [r for r in rows if any(
                        kw in (r.get('title','') or '').lower()
                        or kw in (r.get('site_name','') or '').lower()
                        or kw in (r.get('page_url','') or '').lower()
                        for kw in _include_kw_list
                    )]


                # Apply industry filter — 直接用 gov_raw.industry 列(英文id: power/cpi/pharma...)
                # 历史 bug: 曾从 daily_crawl_config.json 读 site_name→industry 映射, 但 config 无 industry 字段 → 全空 → 选任何行业都 0
                # 行业过滤已下沉到 SQL（见各查询分支），不再对当前页内存过滤 —— 修复"选行业只有2条/0条"
                has_more = len(rows) > per

                if has_more:

                    rows = rows[:per]

                db.close()

        elif db_type == 'eia':

            has_more, rows = False, []

            if os.path.exists(EIA_DB_PATH):

                db = sqlite3.connect(EIA_DB_PATH); db.row_factory = sqlite3.Row

                c = db.cursor()

                fetch_size = per + 1

                offset = (page-1)*per

                if q:

                    inc, exc = parse_google_query(q)

                    fts_parts = []

                    for t in inc:

                        fts_parts.append(t)

                    for t in exc:

                        fts_parts.append('-' + t)

                    if fts_parts:

                        fts_q = ' AND '.join(fts_parts)

                        offset = (page-1)*per

                        c.execute("SELECT p.* FROM projects p JOIN projects_fts f ON p.id = f.rowid WHERE projects_fts MATCH ? ORDER BY rank LIMIT ? OFFSET ?", (sanitize_fts_query(fts_q), fetch_size, offset))

                    else:

                        rows = []

                else:

                    offset = (page-1)*per

                    c.execute("SELECT * FROM projects ORDER BY pub_date DESC, id DESC LIMIT ? OFFSET ?", (fetch_size, offset))

                rows = [dict(r) for r in c.fetchall()]; db.close()

                has_more = len(rows) > per

                if has_more:

                    rows = rows[:per]

        else:

            search_fn = znlh_search if db_type == 'znlh' else zc_search

            combined_q = q

            if q2: combined_q = combined_q + (' ' if combined_q else '') + q2

            if q3: combined_q = combined_q + (' ' if combined_q else '') + q3

            total, rows = search_fn(combined_q, page, per, daterange)

            # znlh/zc 保持原样（这些不频繁，暂时不动）

            has_more = False

        # has_more 用于"加载更多"按钮

        def qp_extra():

            parts = []

            if q: parts.append('q='+urllib.parse.quote(q))

            if q2: parts.append('q2='+urllib.parse.quote(q2))

            if q3: parts.append('q3='+urllib.parse.quote(q3))

            if daterange != 'all': parts.append('range='+daterange)
            if industry_filter: parts.append('industry='+urllib.parse.quote(industry_filter))

            return '&'.join(parts)



        CSS = CSS_STYLE + """

.result-item .dl{font-size:13px;color:#5f6368;margin-top:2px}

"""

        h = '<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>'+title+' - BJIIR项目库</title>'

        h += '<style>' + CSS + '</style></head><body>'

        h += '<div class="container">'



        # Nav (new style)

        h += '<div class="nav">'

        h += '<a href="/" class="nav-logo">BJIIR</a>'

        h += '<div class="nav-links">'

        for nav_id, href_path, nav_label in [('crawler', '/crawler/', '\u73af\u8bc4\u516c\u793a'), ('zc', '/zc/', '\u4e2d\u7b56\u5927\u6570\u636e'), ('cceup', '/ccpc', '\u4e2d\u9879\u7f51'), ('contact', '/contact', '\u8054\u7cfb\u4ebaDB'), ('originals', '/originals/', '\u539f\u59cb\u6570\u636e')]:

            act = ' class="act"' if nav_id == db_type else ''

            h += '<a href="'+href_path+'"'+act+'>'+nav_label+'</a>'

        h += '</div>'

        h += nav_right_html(self.username)

        h += '</div>'



        # Search section (new style)

        h += '<div class="search-section">'

        main_form_action = '/'+db_type+'/'

        h += '<form class="search-box" method="get" action="'+main_form_action+'">'

        h += '<input name="q" placeholder="\u641c\u7d22\u9879\u76ee\u7f16\u53f7\u3001\u540d\u79f0\u3001\u884c\u4e1a\u3001\u5730\u533a\u2026" value="'+esc(q)+'">'

        h += '<button>\u641c\u7d22</button></form></div>'



        # Time range + extra fields

        sel_opts = {'all': '\u5168\u90e8\u65f6\u95f4', '1d': '\u8fc7\u53bb24\u5c0f\u65f6', '7d': '\u8fc7\u53bb\u4e00\u5468', '30d': '\u8fc7\u53bb\u4e00\u4e2a\u6708'}
        # Time range + extra fields

        sel_opts = {'all': '\u5168\u90e8\u65f6\u95f4', '1d': '\u8fc7\u53bb24\u5c0f\u65f6', '7d': '\u8fc7\u53bb\u4e00\u5468', '30d': '\u8fc7\u53bb\u4e00\u4e2a\u6708'}

        # 日期标签：一行全暴露（chips）
        h += '<div class="search-tools"><div class="date-bar" style="display:flex;flex-wrap:wrap;align-items:center;gap:6px;margin-bottom:8px">'
        h += '<span style="font-size:13px;color:#333;margin-right:4px">时间</span>'
        for dv, dl in sel_opts.items():
            _cls = 'date-chip act' if daterange == dv else 'date-chip'
            h += f'<a class="{_cls}" href="?' + '&'.join(f'{k}={urllib.parse.quote(str(v))}' for k, v in [('q',q),('q2',q2),('q3',q3)] if v) + (f'&range={dv}' if dv != 'all' else '') + (f'&industry={urllib.parse.quote(industry_filter)}' if industry_filter else '') + '">' + dl + '</a>'
        h += '</div>'

        # 行业标签：一行全暴露（chips）
        h += '<div style="display:flex;flex-wrap:wrap;align-items:center;gap:6px;margin-bottom:8px">'
        h += '<span style="font-size:13px;color:#333;margin-right:4px">行业</span>'
        # 全部行业
        _cls = 'date-chip act' if not industry_filter else 'date-chip'
        h += f'<a class="{_cls}" href="?' + '&'.join(f'{k}={urllib.parse.quote(str(v))}' for k, v in [('q',q),('q2',q2),('q3',q3)] if v) + (f'&range={daterange}' if daterange != 'all' else '') + '">全行业</a>'
        for _ind in INDUSTRIES:
            _cls = 'date-chip act' if industry_filter == _ind['id'] else 'date-chip'
            h += f'<a class="{_cls}" href="?' + '&'.join(f'{k}={urllib.parse.quote(str(v))}' for k, v in [('q',q),('q2',q2),('q3',q3)] if v) + (f'&range={daterange}' if daterange != 'all' else '') + f'&industry={_ind["id"]}">' + _ind['cn'] + '</a>'
        # other
        _cls = 'date-chip act' if industry_filter == 'other' else 'date-chip'
        h += f'<a class="{_cls}" href="?' + '&'.join(f'{k}={urllib.parse.quote(str(v))}' for k, v in [('q',q),('q2',q2),('q3',q3)] if v) + (f'&range={daterange}' if daterange != 'all' else '') + '&industry=other">其他</a>'
        h += '</div></div>'



        # Total count

        h += '<span class="total-count">' + ('\u5171 ' + str(total) + ' \u6761\u7ed3\u679c \u00b7 ' if total > 0 else '') + '\u7b2c '+str(page)+' \u9875</span>'


        
        # Results
        h += '<div class="results">'

        for idx, r in enumerate(rows):

            pid = str(r.get('id') or r.get('project_id') or r.get('zid') or '')

            name = esc(r.get('project_name','') or r.get('owner_company','') or r.get('title','') or '')

            h += '<div class="result-item">'

            h += '<span class="num">' + str(idx+1) + '</span>'

            h += '<div style="flex:1">'

            h += '<h3><a href="/' + db_type + '/project/' + pid + '">' + name + '</a></h3>'

            h += '<div class="meta">'

            h += '<span>' + esc(r.get('publish_date', '') or '日期未知') + '</span>'
            sn = esc(r.get('site_name', '') or '')
            if sn:
                h += '<span class="source">来源：' + sn + '</span>'

            if db_type == 'crawler' and r.get('title'):
                _oid, _olevel = find_originals_id(r.get('title'))
                if _oid:
                    _olbl, _otype = _orig_id_label(_oid)
                    if _olbl:
                        _ocls = 'orig-id' if _otype == 'proj' else 'orig-id comp'
                        _otip = 'originals 项目级匹配 (Project ID)' if _otype == 'proj' else 'originals 公司级匹配 (Plant ID)'
                        h += f'<span class="{_ocls}" title="{_otip}">{_olbl}: {esc(str(_oid))}</span>'

            if r.get('phase'): h += '<span class="bdg bdg-crawler">'+esc(r['phase'])+'</span>'

            if r.get('total_investment'): h += '<span>'+esc(r['total_investment'])+'</span>'

            elif r.get('budget'): h += '<span>'+esc(r['budget'])+'</span>'

            if r.get('province'): h += '<span>'+esc(r['province'])+('/'+esc(r['city']) if r.get('city') else '')+'</span>'

            h += '</div>'


            h += '</div></div>'



        h += '</div>'
        # Pagination
        if total > per:
            total_pages = max(1, (total + per - 1) // per)
            pag_params = {}
            if q: pag_params['q'] = q
            if q2: pag_params['q2'] = q2
            if q3: pag_params['q3'] = q3
            if daterange != 'all': pag_params['range'] = daterange
            if industry_filter: pag_params['industry'] = industry_filter
            h += self.paginate(page, total_pages, '/'+db_type+'/', pag_params, total_items=total, per=per)

        h += '</div></div></body></html>'

        self.send_response_html(200, h)



    def handle_db_detail(self, db_type, path):


        pid = path.split('/')[-1]

        _track_project_visit(db_type, pid, getattr(self, 'username', 'anonymous'))

        pv_all_visitors = []

        pv_total = 0

        if pid:

            try:

                dbt = sqlite3.connect(SESSION_DB)

                dbt.row_factory = sqlite3.Row

                c = dbt.execute('SELECT COUNT(*) FROM project_visits WHERE project_type=? AND project_id=?',

                                (db_type, str(pid)))

                pv_total = c.fetchone()[0]

                c2 = dbt.execute('SELECT DISTINCT username, visited_at FROM project_visits WHERE project_type=? AND project_id=? ORDER BY visited_at DESC LIMIT 20',

                                 (db_type, str(pid)))

                pv_all_visitors = [dict(r) for r in c2]

                dbt.close()

            except:

                pass

        if db_type == 'crawler':

            p = crawler_detail(pid)

            names = {'crawler':'EIA'}

        elif db_type == 'eia':

            p = None

            if os.path.exists(EIA_DB_PATH):

                db = sqlite3.connect(EIA_DB_PATH); db.row_factory = sqlite3.Row

                c = db.execute("SELECT * FROM projects WHERE id=?", (pid,))

                p = c.fetchone()

                db.close()

            names = {'eia':'EIA项目库'}

            if p: p = dict(p)

        else:

            detail_fn = znlh_detail if db_type == 'znlh' else zc_detail

            names = {'znlh':'中能联合','zc':'中策大数据'}

            p = detail_fn(pid)

        if not p:

            self.send_response_html(404, '<html><body><h1>404</h1></body></html>'); return

        # 2026-09-11：页签标题改用项目名 —— 原来只有路由名，浏览器标签显示「EIA」/
        # 「中策大数据」这种与内容无关的字样（ccpc 页统一时已用项目名，这里补齐）。
        # 取不到项目名再回落到路由名。
        title = esc(p.get('project_name') or p.get('title') or names.get(db_type, db_type))

        # 2026-09-10: ?frag=1 → 只返回正文卡片片段（列表页右侧抽屉复用，避免二次请求整页）
        _frag = getattr(self, '_detail_frag', False) or ('frag=1' in (self.path or '')) or ('frag=1' in (path or ''))
        _uname = getattr(self, 'username', '') or ''
        h = ''
        if not _frag:
            # 2026-09-11：补 viewport —— 原先只有 charset，手机端按 980px 渲染再整体缩小，
            # 详情页字变得极小（v2 列表页一直有这行，这里漏了）。
            h = ('<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8">'
                 '<meta name="viewport" content="width=device-width, initial-scale=1">'
                 '<title>' + title + '</title>')
            h += '<style>' + PAGE_SHELL_CSS + DETAIL_CSS + '</style></head><body>'
            h += '<div class="wrap">'
            h += '<div class="bk-bar"><a class="bk" href="/' + db_type + '/" onclick="history.back();return false">← 返回列表</a>'
            h += nav_right_html(_uname) + '</div>'
        h += render_project_detail_body(db_type, p, pv_total, _uname)
        if not _frag:
            h += '</div></body></html>'

        self.send_response_html(200, h)



    # ─── 搜索页面（gov FTS5 + categories）───



    # ─── Google-style pagination ───

    def paginate(self, page, total_pages, base_url, params, total_items=None, per=None):
        """分页条（2026-09-10 视觉重做：白色卡片 + 底部留白 + 页码信息）。

        params: 查询参数字典（不含 page）
        total_items: 可选，总条数（显示「显示 X-Y · 共 N 条」）
        per: 可选，每页条数
        """
        if total_pages <= 1:
            return ''

        def url(p):
            all_params = {**params, 'page': str(p)}
            qs = '&'.join(f'{k}={urllib.parse.quote(str(v))}' for k, v in all_params.items() if v)
            return base_url + '?' + qs if qs else base_url

        html = PAGER_CSS + '<div class="pagination">'
        _info = '第 <b>%d</b> / %d 页' % (page, total_pages)
        if total_items:
            _ps = per or 10
            _from = (page - 1) * _ps + 1
            _to = min(page * _ps, total_items)
            _info += ' · 显示 %d-%d · 共 %d 条' % (_from, _to, total_items)
        html += '<span class="page-info">' + _info + '</span>'
        if page > 1:
            html += f'<a href="{url(page-1)}" class="page-btn">‹ 上一页</a>'
        pages_set = {1, total_pages}
        if total_pages > 1:
            pages_set.add(total_pages - 1)
        for i in range(page - 1, page + 2):
            if 1 <= i <= total_pages:
                pages_set.add(i)
        sorted_pages = sorted(p for p in pages_set if 1 <= p <= total_pages)
        last = 0
        for p in sorted_pages:
            if p - last > 1:
                html += '<span class="page-ellipsis">…</span>'
            active = ' active' if p == page else ''
            html += f'<a href="{url(p)}" class="page-btn{active}">{p}</a>'
            last = p
        if page < total_pages:
            html += f'<a href="{url(page+1)}" class="page-btn">下一页 ›</a>'
        jump_url = url(1)
        jump_base = jump_url.rsplit('page=', 1)[0] + 'page='
        html += f'<span class="page-jump"><input type="number" min="1" max="{total_pages}" value="{page}" onchange="location.href=\'{jump_base}\'+this.value"> 页</span>'
        html += '</div>'
        return html


    def handle_search_page(self, params):

        q = self.get_param(params, 'q')

        q2 = self.get_param(params, 'q2', '').strip()

        q3 = self.get_param(params, 'q3', '').strip()

        sort = self.get_param(params, 'sort', 'date_desc')

        tag = self.get_param(params, 'tag', '')

        sub = self.get_param(params, 'sub', '')

        daterange = self.get_param(params, 'range', 'all')

        # 2026-09-11: 主搜索页**去掉行业分类模块**（用户决定）——多路由下 zc/ccpc 的
        # industry 是自有中文自由文本（不是这 13 个英文 id），主搜索页又是三库合并渲染，
        # 行业筛选在这里既覆盖不全也容易误导。UI chips 已移除；这里固定为 ''，
        # 后续所有 industry 分支自动失效（保留 plumbing，不动 search()/cceup_search() 签名）。
        industry = ''

        page = int(self.get_param(params, 'page', '1'))

        page_size = int(self.get_param(params, 'page_size', '20'))

        if page_size not in (10, 20, 50, 100): page_size = 20

        if sort not in ('relevance', 'date_desc', 'date_asc'): sort = 'date_desc'



        combined_q = q

        if q2: combined_q = combined_q + (' ' if combined_q else '') + q2

        if q3: combined_q = combined_q + (' ' if combined_q else '') + q3



        # Phase tabs: gov-only search

        if tag:

            if tag == '资源网站':

                sub_sources = {'中项网': cceup_search, '中能联合': znlh_search, '中策大数据': zc_search}

                if sub and sub in sub_sources:

                    total, rows = sub_sources[sub](combined_q, page, page_size, daterange, industry)

                else:

                    all_rows = []
                    _sub_true_total = 0

                    for src_name, search_fn in [('中项网', cceup_search), ('中能联合', znlh_search), ('中策大数据', zc_search)]:

                        t, r = search_fn(combined_q, 1, 10000, daterange, industry)
                        _sub_true_total += (t or 0)

                        for rr in r:

                            rr['_source'] = src_name

                        all_rows.extend(r)

                    all_rows.sort(key=lambda x: x.get('publish_date','') or '', reverse=True)

                    total = len(all_rows)

                    offset = (page - 1) * page_size

                    rows = all_rows[offset:offset+page_size]

                total_pages = max(1, (total + page_size - 1) // page_size)

                offset = (page - 1) * page_size

                html_page = self.render_main_page(q, rows, total, False, page, page_size, sort, tag, sub, offset, daterange, q2, q3, industry,
                                                  true_total=(_sub_true_total or total))

                self.send_response_html(200, html_page)

                return

            # Phase tag: gov search filtered by phase

            phase = tag

            if combined_q or daterange != 'all':

                results, total, has_more = search(combined_q, page, page_size, sort, phase, None, daterange, industry)

            else:

                results, total, has_more = [], 0, False

            offset = (page - 1) * page_size

            html_page = self.render_gov_page(q, results, total, has_more, page, page_size, sort, tag, offset, daterange, q2, q3, industry)

            self.send_response_html(200, html_page)

            return



        # Main page (no tag): search all DBs — 中项网 + 中策大数据 + EIA simultaneously

        all_rows = []

        # 2026-09-11: 三源**真实命中总数**（供「共 N 条结果」诚实显示；见 render_main_page 的 true_total）
        gov_total = cceup_total = zc_total = 0
        if combined_q or daterange != 'all' or (industry and industry != ''):

            # EIA gov
            # 行业筛选时需要全量取回（前200会被更新记录挤掉目标行业旧数据）
            # 2026-09-03: gov_entity 实体全称(公司/电话)命中时拉全量, 否则200截断会把正文命中丢光
            _gq = (combined_q or '').strip()
            _gterms = _gq.split()
            _probe = None
            if len(_gterms) == 1 and ' ' not in _gq and not _gq.startswith('-'):
                _t = _gterms[0]
                if contains_chinese(_t) and len(_t) >= 3:
                    _probe = ('company', _t)
                else:
                    _dig = re.sub(r'\D', '', _t)
                    if 7 <= len(_dig) <= 13:
                        _probe = ('phone', _dig)
            _ecap = 0
            if _probe:
                try:
                    _dbp = get_db()
                    _cp = _dbp.cursor()
                    _pc = ['e.etype = ?', 'e.value = ?']
                    _pp = [_probe[0], _probe[1]]
                    if daterange and daterange != 'all':
                        _days = {'1d': 1, '7d': 7, '30d': 30}[daterange]
                        _pc.append("(r.publish_date GLOB '20[0-9][0-9]-[0-9][0-9]-[0-9][0-9]*' AND SUBSTR(r.publish_date,1,10) >= date('now', '-%d days'))" % _days)
                    if industry and industry != 'all':
                        _pc.append('r.industry = ?')
                        _pp.append(industry)
                    _cp.execute('SELECT COUNT(*) FROM gov_entity e JOIN gov_raw r ON r.id = e.doc_id WHERE ' + ' AND '.join(_pc), _pp)
                    _ecap = _cp.fetchone()[0]
                    _dbp.close()
                except Exception:
                    _ecap = 0
            _gov_fetch = 10000 if (industry and industry != '') else (_ecap if _ecap else 200)
            gov_results, gov_total, _ = search(combined_q, 1, _gov_fetch, 'date_desc', None, None, daterange, industry)

            for rr in gov_results:

                rr['_source'] = 'EIA'

                # 实时打标（若 industry 列为空/other 则用标题推断并写回）
                if not rr.get('industry') or rr['industry'] == 'other':
                    rr['_industry'] = classify_industry(rr.get('title',''))
                else:
                    rr['_industry'] = rr['industry']

            all_rows.extend(gov_results)

            # 中项网

            cceup_total, cceup_rows = cceup_search(combined_q, 1, 200, daterange, industry)

            for rr in cceup_rows:

                rr['_source'] = '中项网'

                rr['_industry'] = classify_industry(rr.get('project_name',''))

            all_rows.extend(cceup_rows)

            # 中策大数据

            zc_total, zc_rows = zc_search(combined_q, 1, 200, daterange, industry)

            for rr in zc_rows:

                rr['_source'] = '中策大数据'

                rr['_industry'] = classify_industry(rr.get('project_name',''))

            all_rows.extend(zc_rows)

        # （原 industry 二次过滤已移除：主搜索页不再有行业分类模块）

        all_rows.sort(key=lambda x: x.get('publish_date','') or '', reverse=True)

        has_more = len(all_rows) > page_size

        offset = (page - 1) * page_size

        rows = all_rows[offset:offset+page_size]

        if not rows and offset > 0:

            # 如果请求的页超出实际数据，回退到第一页

            page = 1

            offset = 0

            rows = all_rows[:page_size]

        # 2026-09-11 诚实总数：三个数据源各有取回上限（EIA=_gov_fetch / 中项网 200 / 中策 200），
        # 原来把"取回条数"当总数显示 —— 实测搜「环境影响评价」显示 201，而 EIA 真实命中 128,685。
        _true_total = (gov_total or 0) + (cceup_total or 0) + (zc_total or 0)
        html_page = self.render_main_page(q, rows, len(all_rows), has_more, page, page_size, sort, '', '', offset, daterange, q2, q3, industry,
                                          true_total=_true_total)

        self.send_response_html(200, html_page)

    def render_gov_page(self, query, results, total, has_more, page, page_size, sort, tag, offset, daterange, q2='', q3='', industry=''):

        def link_params(overrides):

            p = {'q': query, 'q2': q2, 'q3': q3, 'sort': sort, 'tag': tag or '', 'range': daterange, 'page': str(page), 'page_size': str(page_size)}

            if industry: p['industry'] = industry

            p.update({k: str(v) for k, v in overrides.items() if v})

            return '&'.join(f'{k}={urllib.parse.quote(v)}' for k, v in p.items() if v)



        result_rows = ''

        for idx, r in enumerate(results):

            cat_badge = f'<span class="cat-badge">{esc(r.get("category",""))}</span>' if r.get('category') else ''

            src_url = esc(r.get("page_url","") or r.get("source_url","") or '')

            tags_html = ' · '.join(PHASE_MAP.get(t,{}).get("label",t) for t in (r.get("tags") or [])) if r.get("tags") else ''

            result_rows += '<div class="result-item">'

            result_rows += f'<span class="result-num">{offset + idx + 1}</span>'

            result_rows += f'<h3><a href="/detail?id={r.get("id",0)}">{esc(r.get("title",""))}</a></h3>'

            result_rows += '<div class="meta">' + cat_badge


            if tags_html: result_rows += f'<span class="tags">{tags_html}</span>'

            result_rows += f'<span class="date">{esc(r.get("publish_date","") or "日期未知")}</span>'
            site_name = esc(r.get("site_name","") or "")
            if site_name:
                result_rows += f'<span class="source">来源：' + site_name + '</span>'

            if r.get("title"):
                _oid, _olevel = find_originals_id(r.get("title"))
                if _oid:
                    _olbl, _otype = _orig_id_label(_oid)
                    if _olbl:
                        _ocls = 'orig-id' if _otype == 'proj' else 'orig-id comp'
                        _otip = 'originals 项目级匹配 (Project ID)' if _otype == 'proj' else 'originals 公司级匹配 (Plant ID)'
                        result_rows += f'<span class="{_ocls}" title="{_otip}">{_olbl}: {esc(str(_oid))}</span>'

            result_rows += f'<span class="invest">\U0001f4b0 {extract_investment(r.get("summary",""))}</span>'

            result_rows += f'<a href="{src_url}" class="source-link" target="_blank">\U0001f517 \u539f\u6587</a>'

            result_rows += '</div>'


        if not results and query: result_rows = '<div class="no-results">\U0001f615 \u6ca1\u6709\u627e\u5230\u76f8\u5173\u7ed3\u679c</div>'

        elif not results and not query: result_rows = '<div class="no-results">\u8f93\u5165\u5173\u952e\u8bcd\u5f00\u59cb\u641c\u7d22</div>'



        sel_all = ' selected' if daterange == 'all' else ''

        sel_1d = ' selected' if daterange == '1d' else ''

        sel_7d = ' selected' if daterange == '7d' else ''

        sel_30d = ' selected' if daterange == '30d' else ''

        sel_opts = {'all': '全部时间', '1d': '过去24小时', '7d': '过去一周', '30d': '过去一个月'}

        # 日期标签：一行全暴露（chips）
        def date_chip_url(dv):
            p = dict(params_dict)
            p['range'] = dv if dv != 'all' else ''
            p['page'] = '1'
            return '/?' + urllib.parse.urlencode({k: v for k, v in p.items() if v})

        params_dict = {'q': query, 'sort': sort, 'tag': tag or '', 'range': daterange, 'q2': q2, 'q3': q3}

        if industry: params_dict['industry'] = industry

        date_bar = '<div style="display:flex;flex-wrap:wrap;align-items:center;gap:6px;margin-bottom:8px"><span style="font-size:13px;color:#333;margin-right:4px">时间</span>'
        for dv, dl in sel_opts.items():
            _cls = 'date-chip act' if daterange == dv else 'date-chip'
            date_bar += f'<a class="{_cls}" href="{date_chip_url(dv)}">{dl}</a>'
        date_bar += '</div>'


        # Numbered pagination instead of load-more
        total_pages = max(1, (total + page_size - 1) // page_size) if total > 0 else 1
        pagination = self.paginate(page, total_pages, '/', params_dict, total_items=total, per=page_size)



        # 2026-09-11: 主搜索页行业分类模块已移除。变量保留为空串 ——
        # 模板里的插值点不动，零结构改动（改插值点要动多行字符串拼接，风险更高）。
        refine_html = ''



        CSS = CSS_STYLE + """

.pagination{text-align:center;margin:20px 0;display:flex;justify-content:center;gap:4px;flex-wrap:wrap}.page-btn{display:inline-block;padding:8px 14px;background:#fff;border:1px solid #dadce0;border-radius:6px;font-size:14px;color:#1a73e8;text-decoration:none;transition:all .15s}.page-btn:hover{background:#e8f0fe;border-color:#1a73e8}.page-btn.active{background:#1a73e8;color:#fff;border-color:#1a73e8}.page-ellipsis{padding:8px 4px;color:#999}.page-jump{display:inline-flex;align-items:center;gap:4px;margin-left:8px}.page-jump input{width:50px;padding:4px 6px;border:1px solid #dadce0;border-radius:4px;text-align:center;font-size:13px}
.meta .source{color:#0d652d;font-size:12px;margin-left:8px}

"""

        return '''<!DOCTYPE html>\n<html lang="zh-CN">\n<head>\n    <meta charset="UTF-8">\n    <meta name="viewport" content="width=device-width, initial-scale=1.0">\n    <title>''' + (esc(query) if query else '\u9879\u76ee\u4fe1\u606f\u68c0\u7d22') + '''</title>\n    <style>''' + CSS + '''</style>\n</head>\n<body>\n    <div class="container">\n        <div class="nav">

            <a href="/" class="nav-logo">BJIIR</a>

            <div class="nav-links">

                <a href="/crawler/" class="act">环评公示</a>

                <a href="/zc/">中策大数据</a>

                <a href="/ccpc">中项网</a>

                <a href="/contact">联系人DB</a>
                <a href="/originals/">原始数据</a>

                <a href="/originals/">原始数据</a>

            </div>

            ''' + nav_right_html(self.username) + '''

        </div>

        <div class="search-section">\n            <form action="/" method="get" class="search-box">\n                <input type="text" name="q" placeholder="\u641c\u7d22\u9879\u76ee\u5173\u952e\u8bcd\u3001\u6807\u9898\u3001URL..." value="''' + esc(query) + '''" autofocus>\n                <button type="submit">\u641c\u7d22</button>\n            </form>\n        </div>\n        <div class="search-tools">\n            <div class="date-bar">''' + date_bar + '''</div>\n        </div>\n        \n        <div class="toolbar"><div class="controls">\n            \n            <span class="total-count">共 ''' + str(total) + ''' \u6761\u7ed3\u679c</span>\n        </div></div>

        ''' + refine_html + '''

        <div class="results">''' + result_rows + '''</div>\n        ''' + pagination + '''
    </div>\n</body>\n</html>'''

    def render_main_page(self, query, rows, total, has_more, page, page_size, sort, tag, sub, offset, daterange, q2='', q3='', industry='',
                         true_total=None):

        def link_params(overrides):

            p = {'q': query, 'q2': q2, 'q3': q3, 'sort': sort, 'tag': tag or '', 'sub': sub or '', 'range': daterange, 'page': str(page), 'page_size': str(page_size)}

            if industry: p['industry'] = industry

            p.update({k: str(v) for k, v in overrides.items() if v})

            return '&'.join(f'{k}={urllib.parse.quote(v)}' for k, v in p.items() if v)



        result_rows = ''

        for idx, r in enumerate(rows):

            src = r.get('_source', '')

            pid = r.get('id') or r.get('project_id') or r.get('zid') or ''

            name = esc(r.get('project_name','') or r.get('owner_company','') or r.get('title','') or '')

            result_rows += '<div class="result-item">'

            result_rows += f'<span class="num">{offset + idx + 1}</span>'

            if src == 'EIA':

                detail_link = f'/crawler/project/{pid}'

            elif src == '中策大数据':

                detail_link = f'/zc/project/{pid}'

            else:

                detail_link = f'/ccpc/project/{pid}'

            result_rows += f'<h3><a href="{detail_link}">{name}</a></h3>'

            result_rows += '<div class="meta">'

            if src: result_rows += f'<span class="bdg bdg-crawler">{esc(src)}</span>'

            if r.get('budget'): result_rows += f'<span>&#x1f4b0; {esc(str(r["budget"]))}</span>'

            if r.get('province'): result_rows += f'<span>&#x1f4cd; {esc(r["province"])}{"/"+esc(r["city"]) if r.get("city") else ""}</span>'

            result_rows += f'<span>&#x1f4c5; {esc(r.get("publish_date","") or "日期未知")}</span>'

            result_rows += '</div></div>'

        if not result_rows:

            result_rows = '<div class="no-results">&#x1f615; 没有找到相关结果</div>'



        sel_opts = {'all': '全部时间', '1d': '过去24小时', '7d': '过去一周', '30d': '过去一个月'}

        params_dict = {'q': query, 'sort': sort, 'tag': tag or '', 'range': daterange, 'q2': q2, 'q3': q3}

        if industry: params_dict['industry'] = industry

        # 日期标签：一行全暴露（chips）
        def date_chip_url(dv):
            p = dict(params_dict)
            p['range'] = dv if dv != 'all' else ''
            p['page'] = '1'
            return '/?' + urllib.parse.urlencode({k: v for k, v in p.items() if v})

        date_bar = '<div style="display:flex;flex-wrap:wrap;align-items:center;gap:6px;margin-bottom:8px"><span style="font-size:13px;color:#333;margin-right:4px">时间</span>'
        for dv, dl in sel_opts.items():
            _cls = 'date-chip act' if daterange == dv else 'date-chip'
            date_bar += f'<a class="{_cls}" href="{date_chip_url(dv)}">{dl}</a>'
        date_bar += '</div>'

        # Numbered pagination instead of load-more
        total_pages = max(1, (total + page_size - 1) // page_size) if total > 0 else 1
        pagination = self.paginate(page, total_pages, '/', params_dict, total_items=total, per=page_size)

        # 2026-09-11: 主搜索页行业分类模块已移除（见 handle_search_page 顶部注释）。
        # 变量保留为空串，模板插值点 {industry_chips} 不动。
        industry_chips = ''

        CSS = CSS_STYLE + """

.pagination{text-align:center;margin:20px 0;display:flex;justify-content:center;gap:4px;flex-wrap:wrap}.page-btn{display:inline-block;padding:8px 14px;background:#fff;border:1px solid #dadce0;border-radius:6px;font-size:14px;color:#1a73e8;text-decoration:none;transition:all .15s}.page-btn:hover{background:#e8f0fe;border-color:#1a73e8}.page-btn.active{background:#1a73e8;color:#fff;border-color:#1a73e8}.page-ellipsis{padding:8px 4px;color:#999}.page-jump{display:inline-flex;align-items:center;gap:4px;margin-left:8px}.page-jump input{width:50px;padding:4px 6px;border:1px solid #dadce0;border-radius:4px;text-align:center;font-size:13px}
.total-count b{color:#1a73e8}
.cap-note{color:#b26a00}
.cap-more{color:#1a73e8;font-weight:500;margin-left:2px}
.cap-more:hover{text-decoration:underline}

"""

        # 2026-09-11: 诚实总数文案。true_total = 三源真实命中之和（可能远大于取回的 total）。
        # 一旦被截断，就明说"仅展示最新的 M 条"，并给出进 /crawler/ 看全部的入口
        # （v2 列表是游标分页，实测 0.001~0.8s，兜得住任意结果量）。
        _fmt = lambda n: format(int(n or 0), ',')
        _capped = bool(true_total and true_total > total)
        if _capped:
            _total_html = ('共 <b>%s</b> 条结果 <span class="cap-note">（仅展示最新的 %s 条）</span>'
                           ' <a class="cap-more" href="/crawler/?q=%s%s">在环评公示里看全部 →</a>'
                           % (_fmt(true_total), _fmt(total), urllib.parse.quote(query or ''),
                              (('&range=' + daterange) if (daterange and daterange != 'all') else '')))
        else:
            _total_html = '共 <b>%s</b> 条结果' % _fmt(total)

        is_home = not query and not q2 and not q3 and not industry

        if is_home:

            return f'''<!DOCTYPE html>

<html lang="zh-CN">

<head>

    <meta charset="UTF-8">

    <meta name="viewport" content="width=device-width, initial-scale=1.0">

    <title>BJIIR项目库</title>

    <style>{CSS}</style>{CMK_CSS}

</head>

<body>

    <div class="container">

        <div class="topbar">{nav_right_html(self.username, True)}</div>

        <div class="home-wrap">

            <div class="home-logo">BJIIR项目库</div>

            <div class="home-sub">环保项目信息综合搜索</div>

            <div class="home-search">

                <form action="/" method="get" class="search-box">

                    <input type="text" name="q" placeholder="搜索项目关键词、标题..." value="" autofocus>

                    <button type="submit">搜索</button>

                </form>

                <div class="cmk-tip">按 ⌘K 快速搜索</div>

            </div>

            <div class="home-btns">

                <a href="/crawler/">&#x1f50d; 环评公示</a>

                <a href="/zc/">&#x1f4ca; 中策大数据</a>

                <a href="/ccpc">&#x1f4c4; 中项网</a>

                <a href="/contact">&#x1f464; 联系人DB</a>
                <a href="/originals/">&#x1f4cb; 原始数据</a>

            </div>

        </div>

    </div>

{CMK_TAIL}

</body>

</html>'''

        else:

            return f'''<!DOCTYPE html>

<html lang="zh-CN">

<head>

    <meta charset="UTF-8">

    <meta name="viewport" content="width=device-width, initial-scale=1.0">

    <title>{esc(query)} - BJIIR项目库</title>

    <style>{CSS}</style>{CMK_CSS}

    <style>
    /* 搜索结果页容器化（2026-09-10 用户需求：搜索结果页需要容器） */
    body{{background:#f1f3f5}}
    .container{{max-width:1040px}}
    .results{{background:#fff;border:1px solid #e9ecef;border-radius:14px;
      padding:6px 20px 14px;box-shadow:0 1px 3px rgba(0,0,0,.05);margin-top:12px}}
    .result-item:last-child{{border-bottom:none}}
    </style>

</head>

<body>

    <div class="container">

        <div class="nav">

            <a href="/" class="nav-logo">BJIIR</a>

            <div class="nav-links">

                <a href="/crawler/">环评公示</a>

                <a href="/zc/">中策大数据</a>

                <a href="/ccpc">中项网</a>

                <a href="/contact">联系人DB</a>

            </div>

            {nav_right_html(self.username)}

        </div>

        <div class="search-section">

            <form action="/" method="get" class="search-box">

                <input type="text" name="q" placeholder="搜索项目关键词、标题..." value="{esc(query)}" autofocus>

                <button type="submit">搜索</button>

            </form>

            <div class="cmk-tip">按 ⌘K 快速搜索</div>

        </div>

        <div class="toolbar"><div class="controls">

            <span class="total-count">{_total_html}</span>

            <div class="date-bar">{date_bar}</div>

        </div></div>

        {industry_chips}

        <div class="results">{result_rows}</div>

        {pagination}

    </div>

{CMK_TAIL}

</body>

</html>'''



    def render_cceup_page(self, title, category, q, rows, total, page, page_size, offset):

        tp = max(1, (total+page_size-1)//page_size)

        qp = '&q='+urllib.parse.quote(q) if q else ''

        h = '<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>'+title+'</title>'

        h += '<style>*{margin:0;padding:0;box-sizing:border-box}body{font-family:-apple-system,"Microsoft YaHei",sans-serif;background:#f5f5f5;color:#333}.w{max-width:1000px;margin:0 auto;padding:18px 20px 56px}.nav{display:flex;gap:3px;margin-bottom:20px;white-space:nowrap;flex-wrap:nowrap;align-items:center}.nav a{flex:1;text-align:center;padding:6px 10px;background:#fff;border:1px solid #ddd;border-radius:6px;text-decoration:none;color:#555;font-size:13px;flex-shrink:0;min-width:0;font-weight:500}.nav a.act{background:#4361ee;color:#fff;border-color:#4361ee}h1{font-size:22px;margin-bottom:10px}.sb{display:flex;gap:10px;margin-bottom:20px}.sb input{flex:1;padding:10px 15px;border:1px solid #ddd;border-radius:6px;font-size:15px}.sb button{padding:10px 24px;background:#4361ee;color:#fff;border:none;border-radius:6px;cursor:pointer}.st{color:#666;font-size:14px;margin-bottom:15px}.cd{background:#fff;border-radius:8px;padding:16px 20px;margin-bottom:10px;box-shadow:0 1px 3px rgba(0,0,0,.08);cursor:pointer}.cd:hover{box-shadow:0 2px 8px rgba(0,0,0,.12)}.tt{font-size:16px;font-weight:600;color:#1a1a2e;margin-bottom:6px}.mt{font-size:13px;color:#666;display:flex;gap:16px;flex-wrap:wrap}.mt>*{margin-right:12px}.page-btn{display:inline-block;padding:6px 14px;background:#fff;border:1px solid #ddd;border-radius:6px;text-decoration:none;color:#333;font-size:14px;margin:0 2px}.page-btn.active{background:#4361ee;color:#fff;border-color:#4361ee}.page-ellipsis{display:inline-block;padding:6px 4px;color:#999;font-size:14px}.page-jump{display:inline-flex;align-items:center;gap:4px;margin-left:8px;font-size:13px;color:#666}.page-jump input{width:50px;padding:6px 8px;border:1px solid #ddd;border-radius:6px;font-size:13px;text-align:center;outline:none}.page-jump input:focus{border-color:#2563eb}</style></head><body><div class="w">'

        h += '<div class=\"nav\">'

        for nav_id, href_path, nav_label in [('zc', '/zc/', '中策大数据'), ('cceup', '/ccpc', '中项网'), ('contact', '/contact', '联系人DB'), ('originals', '/originals/', '原始数据'), ('eia', '/eia/', 'EIA项目库'), ('crawler', '/crawler/', 'EIA')]:

            h += '<a class="nav" href="'+href_path+'">'+nav_label+'</a>'

        h += '<span style="flex:1"></span><span style="font-size:12px;color:#888;white-space:nowrap">'+self.username+'</span>'

        h += '<span style="flex:1"></span><span style="font-size:13px;color:#555;white-space:nowrap">登录用户 '+self.username+'</span><a href="/logout" style="font-size:13px;color:#e53935;text-decoration:none;margin-left:8px">退出</a></div></div>'

        h += '<h1>'+title+'</h1>'

        h += '<form class="sb" method="get" action="/"><input type="hidden" name="category" value="中项网"><input name="q" placeholder="搜索…" value="'+esc(q)+'"><button>搜索</button></form>'

        h += '<div class="st">共 '+str(total)+' 个项目</div>'

        for r in rows:

            pid = str(r.get('id') or r.get('project_id') or r.get('zid') or '')

            name = esc(r.get('project_name','') or r.get('owner_company','') or r.get('title','') or '')

            h += '<div class="cd" onclick="location.href=\'/ccpc/project/'+pid+'\'"><div class="tt">'+name+'</div><div class="mt">'

            

            if r.get('budget'): h += '<span>'+esc(r['budget'])+'</span>'

            if r.get('province'): h += '<span>'+esc(r['province'])+('/'+esc(r['city']) if r.get('city') else '')+'</span>'

            if r.get('publish_date'): h += '<span>'+esc(r['publish_date'])+'</span>'

            h += '</div></div>'

        if tp > 1:

            pag_params = {'category': '中项网'}

            if q: pag_params['q'] = q

            h += self.paginate(page, tp, '/', pag_params, total_items=total, per=page_size)

        h += '</div></div></body></html>'

        return h



    # ─── 原有CCEUP页面 ───



    def handle_cceup_page(self, params):

        page = int(self.get_param(params, 'page', '1'))

        page_size = 50

        q = self.get_param(params, 'q', '').strip()

        conn = sqlite3.connect(CONTACT_DB_PATH)

        conn.row_factory = sqlite3.Row

        c = conn.cursor()

        if q:

            c.execute("SELECT COUNT(*) FROM cceup_contacts WHERE company LIKE ? OR contact_name LIKE ? OR phone LIKE ? OR email LIKE ?", (f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%'))

            total = c.fetchone()[0]

            offset = (page - 1) * page_size

            c.execute("SELECT * FROM cceup_contacts WHERE company LIKE ? OR contact_name LIKE ? OR phone LIKE ? OR email LIKE ? ORDER BY id DESC LIMIT ? OFFSET ?", (f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', page_size, offset))

        else:

            c.execute("SELECT COUNT(*) FROM cceup_contacts")

            total = c.fetchone()[0]

            offset = (page - 1) * page_size

            c.execute("SELECT * FROM cceup_contacts ORDER BY id DESC LIMIT ? OFFSET ?", (page_size, offset))

        rows = [dict(r) for r in c.fetchall()]

        conn.close()

        total_pages = max(1, (total + page_size - 1) // page_size)

        self.send_response_html(200, self.render_contact_list(q, rows, total, page, page_size, total_pages, offset))



    def handle_contact_page(self, params):
        # 2026-09-10: /contact 联系人DB 切到统一联系人库 contact.db(contacts 表, contact_lib.py 聚合)
        # 原先读 CONTACT_DB_PATH(=ccpc.db) 的 cceup_contacts —— 保留给 /ccpc 项目浏览使用
        page = int(self.get_param(params, 'page', '1'))

        page_size = 50

        q = self.get_param(params, 'q', '').strip()

        conn = sqlite3.connect(CONTACT_ROUTE_DB)

        conn.row_factory = sqlite3.Row

        c = conn.cursor()

        if q:

            c.execute("SELECT COUNT(*) FROM contacts WHERE company LIKE ? OR contact_name LIKE ? OR phone LIKE ? OR email LIKE ? OR role LIKE ? OR department LIKE ? OR position LIKE ? OR remarks LIKE ? OR landline LIKE ?", (f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%'))

            total = c.fetchone()[0]

            offset = (page - 1) * page_size

            c.execute("SELECT * FROM contacts WHERE company LIKE ? OR contact_name LIKE ? OR phone LIKE ? OR email LIKE ? OR role LIKE ? OR department LIKE ? OR position LIKE ? OR remarks LIKE ? OR landline LIKE ? ORDER BY id DESC LIMIT ? OFFSET ?", (f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', f'%{q}%', page_size, offset))

        else:

            c.execute("SELECT COUNT(*) FROM contacts")

            total = c.fetchone()[0]

            offset = (page - 1) * page_size

            c.execute("SELECT * FROM contacts ORDER BY id DESC LIMIT ? OFFSET ?", (page_size, offset))

        rows = [dict(r) for r in c.fetchall()]

        conn.close()

        total_pages = max(1, (total + page_size - 1) // page_size)

        self.send_response_html(200, self.render_contact_list(q, rows, total, page, page_size, total_pages, offset))



    def render_contact_list(self, q, rows, total, page, page_size, total_pages, offset):

        CSS = CSS_STYLE + '''

.contact-card{border:1px solid #e8e8e8;border-radius:8px;padding:16px;margin-bottom:12px;background:#fafafa;display:flex;align-items:flex-start;gap:16px}

.contact-body{flex:1;min-width:0}

.contact-dial{flex-shrink:0;display:flex;flex-direction:column;align-items:center;justify-content:center;padding:8px}

.contact-dial a{display:flex;align-items:center;justify-content:center;width:48px;height:48px;border-radius:50%;background:#1a73e8;color:#fff;text-decoration:none;font-size:22px}

.contact-dial a:hover{background:#1557b0}

'''

        h = '<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>联系人DB - BJIIR项目库</title><style>'+CSS+'</style></head><body>'

        h += '<div class="container">'



        # Nav (new style)

        h += '<div class="nav">'

        h += '<a href="/" class="nav-logo">BJIIR</a>'

        h += '<div class="nav-links">'

        for nav_id, href_path, nav_label in [('crawler', '/crawler/', '\u73af\u8bc4\u516c\u793a'), ('zc', '/zc/', '\u4e2d\u7b56\u5927\u6570\u636e'), ('cceup', '/ccpc', '\u4e2d\u9879\u7f51'), ('contact', '/contact', '\u8054\u7cfb\u4ebaDB'), ('originals', '/originals/', '\u539f\u59cb\u6570\u636e')]:

            act = ' class="act"' if nav_id == 'contact' else ''

            h += '<a href="'+href_path+'"'+act+'>'+nav_label+'</a>'

        h += '</div>'

        h += nav_right_html(self.username)

        h += '</div>'



        # Search section

        h += '<div class="search-section">'

        h += '<form class="search-box" method="get" action="/contact">'

        h += '<input name="q" placeholder="\u641c\u7d22\u516c\u53f8\u540d\u3001\u8054\u7cfb\u4eba\u3001\u624b\u673a\u53f7\u2026" value="'+esc(q)+'">'

        h += '<button>\u641c\u7d22</button></form></div>'



        h += '<div class="toolbar"><div class="controls">'

        h += '<span class="total-count">\u5171 '+str(total)+' \u6761\u8054\u7cfb\u4eba</span>'

        h += '</div></div>'



        for idx, r in enumerate(rows):

            h += '<div class="contact-card">'

            h += '<div class="contact-body">'

            h += '<div style="font-size:16px;font-weight:600;color:#1a1a2e;margin-bottom:6px">'+esc(r.get('company','') or '')+'</div>'

            h += '<div style="font-size:13px;color:#666;display:flex;flex-wrap:wrap;gap:4px 16px;line-height:1.6">'

            h += '<span><span style="color:#888;margin-right:2px">\u8054\u7cfb\u4eba\uff1a</span>'+esc(r.get('contact_name','') or '')+'</span>'

            h += '<span><span style="color:#888;margin-right:2px">\u624b\u673a\uff1a</span>'+esc(r.get('phone','') or '')+'</span>'

            if r.get('landline'): h += '<span><span style="color:#888;margin-right:2px">\u5ea7\u673a\uff1a</span>'+esc(r['landline'])+'</span>'

            _role = contact_role_display(r.get('role'))

            if _role: h += '<span><span style="color:#888;margin-right:2px">\u89d2\u8272\uff1a</span>'+esc(_role)+'</span>'

            if r.get('department'): h += '<span><span style="color:#888;margin-right:2px">\u90e8\u95e8\uff1a</span>'+esc(r['department'])+'</span>'

            if r.get('position'): h += '<span><span style="color:#888;margin-right:2px">\u804c\u52a1\uff1a</span>'+esc(r['position'])+'</span>'

            if r.get('remarks'): h += '<span><span style="color:#888;margin-right:2px">\u5907\u6ce8\uff1a</span>'+esc(r['remarks'])+'</span>'

            if r.get('email'): h += '<span><span style="color:#888;margin-right:2px">\u90ae\u7bb1\uff1a</span>'+esc(r['email'])+'</span>'

            if r.get('address'): h += '<span><span style="color:#888;margin-right:2px">\u5730\u5740\uff1a</span>'+esc(r['address'])+'</span>'

            h += '</div></div>'

            phone = r.get('phone', '')

            if phone:

                h += '<div class="contact-dial"><a href="tel:'+esc(phone)+'">\u260e</a></div>'

            h += '</div>'



        if total_pages > 1:

            pag_params = {}

            if q: pag_params['q'] = q

            h += self.paginate(page, total_pages, '/contact', pag_params, total_items=total, per=page_size)

        h += '</div></div></body></html>'

        return h



    def handle_cceup_projects(self, params):

        from html import escape

        page = int(self.get_param(params, 'page', '1'))

        q = self.get_param(params, 'q', '').strip()

        q2 = self.get_param(params, 'q2', '').strip()

        q3 = self.get_param(params, 'q3', '').strip()

        daterange = self.get_param(params, 'range', 'all')

        page_size = 50

        combined_q = q

        if q2: combined_q = combined_q + (' ' if combined_q else '') + q2

        if q3: combined_q = combined_q + (' ' if combined_q else '') + q3

        total, rows = cceup_search(combined_q, page, page_size, daterange)

        total_pages = max(1, (total + page_size - 1) // page_size)

        CSS = CSS_STYLE + """

"""

        h = '<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>中项网 - BJIIR项目库</title><style>'+CSS+'</style></head><body>'

        h += '<div class="container">'



        # Nav (new style)

        h += '<div class="nav">'

        h += '<a href="/" class="nav-logo">BJIIR</a>'

        h += '<div class="nav-links">'

        for nav_id, href_path, nav_label in [('crawler', '/crawler/', '\u73af\u8bc4\u516c\u793a'), ('zc', '/zc/', '\u4e2d\u7b56\u5927\u6570\u636e'), ('cceup', '/ccpc', '\u4e2d\u9879\u7f51'), ('contact', '/contact', '\u8054\u7cfb\u4ebaDB'), ('originals', '/originals/', '\u539f\u59cb\u6570\u636e')]:

            act = ' class="act"' if nav_id == 'cceup' else ''

            h += '<a href="'+href_path+'"'+act+'>'+nav_label+'</a>'

        h += '</div>'

        h += nav_right_html(self.username)

        h += '</div>'



        # Search section

        h += '<div class="search-section">'

        h += '<form class="search-box" method="get" action="/ccpc">'

        h += '<input name="q" placeholder="\u641c\u7d22\u4e2d\u9879\u7f51\u9879\u76ee\u2026" value="'+escape(q)+'">'

        h += '<button>\u641c\u7d22</button></form></div>'



        # Time range

        sel_opts = {'all': '\u5168\u90e8\u65f6\u95f4', '1d': '\u8fc7\u53bb24\u5c0f\u65f6', '7d': '\u8fc7\u53bb\u4e00\u5468', '30d': '\u8fc7\u53bb\u4e00\u4e2a\u6708'}

        # 日期标签：一行全暴露（chips）
        def _date_chip_url(dv):
            _p = []
            if q: _p.append('q=' + urllib.parse.quote(q))
            if q2: _p.append('q2=' + urllib.parse.quote(q2))
            if q3: _p.append('q3=' + urllib.parse.quote(q3))
            if dv != 'all': _p.append('range=' + dv)
            return '/ccpc?' + '&'.join(_p)

        h += '<div class="search-tools"><div class="date-bar" style="display:flex;flex-wrap:wrap;align-items:center;gap:6px;margin-bottom:8px"><span style="font-size:13px;color:#333;margin-right:4px">时间</span>'
        for dv, dl in sel_opts.items():
            _cls = 'date-chip act' if daterange == dv else 'date-chip'
            h += f'<a class="{_cls}" href="{_date_chip_url(dv)}">{dl}</a>'
        h += '</div></div>'



        h += '<div class="toolbar"><div class="controls">'

        h += '<span class="total-count">' + ('\u5171 ' + str(total) + ' \u6761\u7ed3\u679c \u00b7 ' if total > 0 else '') + '\u7b2c '+str(page)+' \u9875</span>'

        h += '</div></div>'



        for idx, r in enumerate(rows):

            pid = str(r.get('id') or r.get('project_id') or r.get('zid') or '')

            name = escape(r.get('project_name','') or '')

            h += '<div class="result-item">'

            h += '<span class="num">' + str(idx+1) + '</span>'

            h += '<div style="flex:1">'

            h += '<h3><a href="/ccpc/project/'+pid+'">'+name+'</a></h3>'

            h += '<div class="meta">'

            if r.get('phase'): h += '<span class="bdg bdg-crawler">'+escape(r['phase'])+'</span>'

            if r.get('budget'): h += '<span>'+escape(r['budget'])+'</span>'

            if r.get('province'): h += '<span>'+escape(r['province'])+('/'+escape(r['city']) if r.get('city') else '')+'</span>'

            if r.get('publish_date'): h += '<span>'+escape(r['publish_date'])+'</span>'

            h += '</div></div></div>'
        h += '</div>'

        if total_pages > 1:

            pag_params = {}

            if q: pag_params['q'] = q

            if q2: pag_params['q2'] = q2

            if q3: pag_params['q3'] = q3

            if daterange != 'all': pag_params['range'] = daterange

            h += self.paginate(page, total_pages, '/ccpc', pag_params, total_items=total, per=page_size)

        h += '</div></div></body></html>'

        self.send_response_html(200, h)



    def handle_cceup_project_detail(self, path):
        """中项网(CCPC)项目独立详情页。

        2026-09-11 统一：原实现是另一套老设计（.detail-wrap/.detail-title/
        .detail-card/.detail-grid/.label/.value/.detail-text + 自带一大段 <style>），
        与 v2 详情页/抽屉完全两套视觉，且末尾多出 1 个游离 </div>。
        现改为 PAGE_SHELL_CSS + DETAIL_CSS + render_cceup_detail_body()，
        与 /crawler/project/*、/zc/project/* 一致，并保留老版的全部字段。
        """
        try:
            pid = int((path or '').rstrip('/').split('/')[-1])
        except ValueError:
            self.send_response_html(404, 'Not found'); return
        conn = sqlite3.connect(CONTACT_DB_PATH)
        conn.row_factory = sqlite3.Row
        try:
            r = conn.execute("SELECT * FROM cceup_projects WHERE id=?", (pid,)).fetchone()
            if not r:
                self.send_response_html(404, 'Not found'); return
            p = dict(r)
            contacts = [dict(x) for x in conn.execute(
                "SELECT * FROM cceup_contacts WHERE project_id=? ORDER BY id", (pid,))]
        finally:
            conn.close()
        _uname = getattr(self, 'username', '') or ''
        h = ('<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8">'
             '<meta name="viewport" content="width=device-width, initial-scale=1">'
             '<title>' + esc(p.get('project_name') or '中项网项目') + ' · BJIIR项目库</title>'
             '<style>' + PAGE_SHELL_CSS + DETAIL_CSS + '</style></head><body>'
             '<div class="wrap">'
             '<div class="bk-bar"><a class="bk" href="/ccpc"'
             ' onclick="history.back();return false">← 返回列表</a>'
             + nav_right_html(_uname) + '</div>')
        h += render_cceup_detail_body(p, contacts)
        h += '</div></body></html>'
        self.send_response_html(200, h)



    def handle_detail(self, params):

        record_id = int(self.get_param(params, 'id', '0'))

        record = get_detail(record_id)

        if not record: self.send_response_html(404, 'Not found'); return

        from html import escape

        title = escape(record.get('title', 'Unknown'))

        site = escape(record.get('site') or '')

        cat = record.get('category') or ''

        date = record.get('publish_date') or ''

        status = record.get('status', '已采集')

        summary = record.get('summary', '')

        page_url = record.get('page_url') or ''

        source_url = record.get('source_url') or ''

        tags = record.get('tags') or ''

        if isinstance(tags, str): tags = [t.strip() for t in tags.split(',') if t.strip()]

        cat_html = f'<span class="cat-badge">{escape(cat)}</span>' if cat else ''

        record_id_str = f'<a href="/detail?id={record_id}">#{record_id}</a>'

        summary_html = render_md(summary) if summary else ''
        content_raw = strip_body_noise(record.get('content', ''))   # ⚠️ 剥 <style>/<script>：原样注入会污染全页 CSS
        # Prefer clean HTML content; fall back to summary as markdown
        if content_raw and len(content_raw) > 50 and ('<p>' in content_raw or '<table' in content_raw):
            body_html = content_raw
        elif content_raw and len(content_raw) > 50:
            # Treat as Markdown (pipe tables, etc.) and convert to HTML
            body_html = render_md(content_raw)
        elif summary and len(summary) > 50:
            body_html = summary_html
        else:
            body_html = '<p class="no-summary">正文内容未采集完整，请点击下方原文链接查看完整信息</p>'

        show_source = f'<p class="detail-source-item"><span class="detail-label">📡 来源网站</span><span>{site}</span></p>' if site else ''

        search_q = self.get_param(params, 'q', '')

        search_sort = self.get_param(params, 'sort', 'date_desc')

        search_phase = self.get_param(params, 'phase', '')

        search_category = self.get_param(params, 'category', '')

        search_page = self.get_param(params, 'page', '1')

        search_psize = self.get_param(params, 'page_size', '20')



        CSS = '''body{font-family:-apple-system,"Microsoft YaHei",sans-serif;background:#f5f5f5;margin:0;padding:0;color:#333}

        .container{max-width:800px;margin:0 auto;padding:20px 20px 56px}

        .nav-bar{margin-bottom:12px}.nav-bar a{color:#4361ee;text-decoration:none;font-size:14px}

        .detail-card{background:#fff;border-radius:12px;box-shadow:0 1px 4px rgba(0,0,0,.08);padding:24px}

        .detail-title{font-size:24px;margin-bottom:16px;color:#1a1a2e}

        .detail-meta{display:flex;flex-wrap:wrap;gap:12px;margin-bottom:20px;font-size:13px;color:#666;align-items:center}

        .detail-meta span{background:#f0f0f0;padding:4px 10px;border-radius:4px}

        .detail-status{color:#2e7d32;background:#e8f5e9!important}

        .detail-visits{color:#e65100}

        .detail-body{font-family:-apple-system,BlinkMacSystemFont,"Microsoft YaHei","PingFang SC",sans-serif;font-size:15px;line-height:1.6;color:#333}

        .detail-body p{margin:6px 0}.detail-body table{border-collapse:collapse;width:100%;margin:14px 0;font-size:14px}.detail-body td,.detail-body th{border:1px solid #ddd;padding:8px 12px;text-align:left;vertical-align:top}.detail-body tr:nth-child(even){background:#f9f9f9}

        .detail-source{border-top:1px solid #e5e7eb;padding-top:16px;margin-top:20px;font-size:14px}

        .detail-source-item{margin:8px 0}

        .detail-label{color:#888;margin-right:8px}

        .detail-original-link{color:#4361ee;word-break:break-all}'''

        h = f'''<!DOCTYPE html><html lang="zh-CN"><head><meta charset="UTF-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>{title}</title><style>{CSS}</style></head><body><div class="container">

        <div class="nav-bar"><a href="/?q={urllib.parse.quote(search_q)}&sort={search_sort}&phase={search_phase}&category={search_category}&page={search_page}&page_size={search_psize}" if search_q else '/'">&larr; 返回搜索结果</a></div>

        <div class="detail-card"><h1 class="detail-title">{title}</h1>

        <div class="detail-meta">{cat_html}<span>{site}</span><span>📅 {date or "日期未知"}</span><span class="detail-status">{status or "已采集"}</span><span class="detail-visits">👁 {record.get('visits',0) or 0} 次访问</span></div>

        <div class="detail-body">{body_html}</div>

        <div class="detail-source"><p class="detail-source-item"><span class="detail-label">🔗 原文链接</span><a href="{page_url or source_url}" target="_blank" rel="noopener" class="detail-original-link">{page_url or source_url}</a></p>{show_source}<p class="detail-source-item"><span class="detail-label">数据ID</span>{record_id_str}</p></div></div>

        <script>fetch('/api/visit?id={record_id}');</script></body></html>'''

        self.send_response_html(200, h)



    def handle_visit(self, params):

        record_id = int(self.get_param(params, 'id', '0'))

        if record_id:

            conn = get_db()

            conn.execute("UPDATE gov_raw SET visits = visits + 1 WHERE id=?", (record_id,))

            conn.commit(); conn.close()

        self.send_response_html(200, 'ok')



    def handle_dashboard(self):

        self.send_response_html(200, '<html><body><h1>Dashboard</h1><p>Coming soon</p></body></html>')



    def handle_sources_api(self):

        self.send_json({'sources': []})



    def send_json(self, data):

        b = json.dumps(data, ensure_ascii=False).encode()

        self.send_response(200)

        self.send_header('Content-Type', 'application/json; charset=utf-8')

        self.send_header('Content-Length', str(len(b)))

        self.end_headers()

        self.wfile.write(b)



    def log_message(self, format, *args): pass





import json, sqlite3, os, subprocess, re

CONFIG_PATH = "/root/gov_crawler/daily_crawl_config.json"
SEARCH_DB = "/root/search.db"
GOV_CRAWLER_DIR = "/root/gov_crawler"
ADMIN_EXCLUDE_PATH = "/root/gov_crawler/admin_exclude_keywords.txt"

INDUSTRIES = [
    {"id": "power", "cn": "电力", "en": "Power"},
    {"id": "terminals", "cn": "终端/接收站", "en": "Terminals"},
    {"id": "port", "cn": "港口码头", "en": "Port & Harbor"},
    {"id": "transmission", "cn": "管道输送", "en": "Transmission"},
    {"id": "production", "cn": "生产/开采", "en": "Production"},
    {"id": "altfuel", "cn": "替代燃料", "en": "Alternative Fuel"},
    {"id": "hpi", "cn": "石油炼制", "en": "HPI (Hydrocarbon Processing Industry)"},
    {"id": "cpi", "cn": "化工加工", "en": "CPI (Chemical Processing Industry)"},
    {"id": "metals", "cn": "金属与矿物", "en": "Metals & Minerals"},
    {"id": "pulp", "cn": "制浆/造纸/木材", "en": "Pulp, Paper & Wood"},
    {"id": "food", "cn": "食品与饮料", "en": "Food & Beverage"},
    {"id": "logistics", "cn": "物流", "en": "Logistics"},
    {"id": "manufacturing", "cn": "工业制造", "en": "Industrial Manufacturing"},
    {"id": "pharma", "cn": "制药与生物技术", "en": "Pharmaceutical & Biotech"},
]

def get_industries_json():
    import json as _j
    return _j.dumps(INDUSTRIES, ensure_ascii=False)

INDUSTRY_IDS = [ind["id"] for ind in INDUSTRIES]
INDUSTRY_CN_MAP = {ind["id"]: ind["cn"] for ind in INDUSTRIES}

def ind_cn(industry_id):
    """行业 id → 中文名（详情页显示用）；未知/other 保持原样"""
    if not industry_id:
        return ''
    return INDUSTRY_CN_MAP.get(industry_id, industry_id)



_SCRIPT_URL_CACHE = {}
_SCRIPT_SITE_CACHE = {}

def extract_script_sitename(script_name):
    """Extract SITE_NAME constant from script source to match DB"""
    if not script_name:
        return ""
    for prefix in ["node ", "bash ", "python3 "]:
        if script_name.startswith(prefix):
            script_name = script_name[len(prefix):]
            break
    if script_name in _SCRIPT_SITE_CACHE:
        return _SCRIPT_SITE_CACHE.get(script_name, "")
    script_path = os.path.join(GOV_CRAWLER_DIR, script_name)
    if not os.path.isfile(script_path):
        _SCRIPT_SITE_CACHE[script_name] = ""
        return ""
    try:
        with open(script_path, "r", errors="replace") as f:
            head = f.read(5000)
    except:
        _SCRIPT_SITE_CACHE[script_name] = ""
        return ""
    m = re.search("^SITE_NAME\s*=\s*[\\\"']([^\\\"']+)[\\\"']", head, re.MULTILINE)
    if not m:
        m = re.search("^NAME\s*=\s*[\\\"']([^\\\"']+)[\\\"']", head, re.MULTILINE)
    if not m:
        m = re.search("^SITE\s*=\s*[\\\"']([^\\\"']+)[\\\"']", head, re.MULTILINE)
    sn = m.group(1) if m else ""
    _SCRIPT_SITE_CACHE[script_name] = sn
    return sn

def extract_script_url(script_name):
    if not script_name:
        return ""
    for prefix in ["node ", "bash ", "python3 "]:
        if script_name.startswith(prefix):
            script_name = script_name[len(prefix):]
            break
    if script_name in _SCRIPT_URL_CACHE:
        return _SCRIPT_URL_CACHE.get(script_name, "")
    script_path = os.path.join(GOV_CRAWLER_DIR, script_name)
    if not os.path.isfile(script_path):
        _SCRIPT_URL_CACHE[script_name] = ""
        return ""
    try:
        with open(script_path, "r", errors="replace") as f:
            head = f.read(5000)
    except:
        _SCRIPT_URL_CACHE[script_name] = ""
        return ""
    # Priority 1: LIST_URL (most precise - list page URL)
    m = re.search("(?m)^(?:LIST_URL|list_url|listUrl)\\s*=\\s*[\"\'](https?://[^\"\']+)[\"\']", head)
    # Priority 2: API_URL
    if not m:
        m = re.search("(?m)^(?:API_URL|api_url|apiUrl)\\s*=\\s*[\"\'](https?://[^\"\']+)[\"\']", head)
    # Priority 3: PAGE_URL / SEARCH_URL / BASE_URL
    if not m:
        m = re.search("(?m)^(?:PAGE_URL|SEARCH_URL|search_url|BASE_URL)\\s*=\\s*[\"\'](https?://[^\"\']+)[\"\']", head)
    # Priority 4: BASE / base_url / DOMAIN
    if not m:
        m = re.search("(?m)^(?:BASE|base_url|DOMAIN)\\s*=\\s*[\"\'](https?://[a-zA-Z0-9.-]+\\.[a-zA-Z]{2,}[^\"\']*)[\"\']", head)
    # Priority 5: domain from docstring
    if not m:
        m = re.search(r"(https?://[a-zA-Z0-9.-]+\.[a-zA-Z]{2,})", head)
    url = m.group(1).rstrip("/") if m else ""
    _SCRIPT_URL_CACHE[script_name] = url
    return url

def set_crawler_industry(name, industry_id):
    """Set industry for a crawler in config"""
    cfg = json.load(open(CONFIG_PATH))
    if isinstance(cfg, list):
        cfg = {"crawlers": cfg}
    if industry_id and industry_id not in INDUSTRY_IDS:
        return {"success": False, "error": f"Unknown industry: {industry_id}"}
    for c in cfg["crawlers"]:
        if c["name"] == name:
            if industry_id:
                c["industry"] = industry_id
            else:
                c.pop("industry", None)
            json.dump(cfg, open(CONFIG_PATH, "w"), ensure_ascii=False, indent=2)
            return {"success": True, "name": name, "industry": industry_id or ""}
    return {"success": False, "error": f"Crawler \"{name}\" not found"}

def get_crawler_stats():
    """获取所有爬虫的状态数据"""
    cfg = json.load(open(CONFIG_PATH))
    if isinstance(cfg, list):
        cfg = {"crawlers": cfg}
    crawlers = cfg['crawlers']
    db = sqlite3.connect(SEARCH_DB)
    db.row_factory = sqlite3.Row
    # Query 1: GROUP BY site_name (fast, uses index)
    rows = db.execute("""
        SELECT site_name,
               COUNT(*) as total,
               MAX(SUBSTR(publish_date,1,10)) as latest_date,
               MIN(SUBSTR(publish_date,1,10)) as oldest_date,
               SUM(CASE WHEN SUBSTR(publish_date,1,10) >= date('now', '-7 days') THEN 1 ELSE 0 END) as week_new,
               SUM(CASE WHEN SUBSTR(publish_date,1,10) >= date('now', '-1 days') THEN 1 ELSE 0 END) as day_new
        FROM gov_raw
        GROUP BY site_name
    """).fetchall()
    db_stats = {r['site_name']: dict(r) for r in rows}
    def normalize(s):
        return s.replace(' ', '').replace('—', '-').replace('–', '-').replace('（', '(').replace('）', ')').replace('　', '').replace(' ', '')
    db_stats_norm = {normalize(k): v for k, v in db_stats.items()}
    # 2026-09-10 提速: 一次批量查询建 script_name -> {site_name} 映射
    # (原实现每个配置各开一次连接、共 3856 次查询 ≈ 8.2s；改走 idx_gov_raw_script_site 覆盖索引 ≈ 0.16s)
    _sn_set = set()
    _sn_to_sites = {}
    try:
        _db_sn = __import__('sqlite3').connect(SEARCH_DB)
        for _r in _db_sn.execute("SELECT DISTINCT script_name, site_name FROM gov_raw WHERE script_name IS NOT NULL AND script_name != ''").fetchall():
            _sn_set.add(_r[0])
            _sn_to_sites.setdefault(_r[0], set()).add(_r[1])
        _db_sn.close()
    except:
        _sn_set = set()
        _sn_to_sites = {}
    db.close()

    # 2026-09-10 提速: run_logs 一次性预载（原实现每个配置各开连接查 3~4 次，2086 配置 ≈ 8s）
    # 今日状态按 script_name / config_name 双键索引；历史各取最近 5 条
    _rl_today = {}
    _rl_hist = {}
    _rl_load_failed = False
    try:
        _conn = __import__('sqlite3').connect(SEARCH_DB)
        _conn.row_factory = __import__('sqlite3').Row
        _today_s = __import__('datetime').datetime.now().strftime('%Y-%m-%d')
        for _r in _conn.execute(
                "SELECT script_name, config_name, status, new_count, elapsed_seconds, error_detail, run_time "
                "FROM run_logs WHERE run_date=?", (_today_s,)).fetchall():
            _d = dict(_r)
            for _k in (_r['script_name'] or '', _r['config_name'] or ''):
                if _k:
                    _rl_today.setdefault(_k, _d)
        for _r in _conn.execute(
                "SELECT script_name, config_name, run_date, status, new_count "
                "FROM run_logs WHERE run_date<=? ORDER BY run_date DESC", (_today_s,)).fetchall():
            _h = {'run_date': _r['run_date'], 'status': _r['status'], 'new_count': _r['new_count']}
            for _k in (_r['script_name'] or '', _r['config_name'] or ''):
                if _k:
                    _lst = _rl_hist.setdefault(_k, [])
                    if len(_lst) < 5:
                        _lst.append(_h)
        _conn.close()
    except Exception:
        _rl_load_failed = True
        _rl_today = {}
        _rl_hist = {}

    result = []
    for c in crawlers:
        name = c.get('name', c.get('site_name', c.get('script', '')))
        script = c.get('script', c.get('command', ''))
        # 唯一标识: script_name字段 -> args里的--script-name -> 脚本文件名
        _ck = c.get('script_name', '')
        if not _ck:
            _a = c.get('args', '')
            if isinstance(_a, list):
                _a = ' '.join(str(x) for x in _a)
            _m = re.search(r'--script-name=(\S+)', _a) if _a else None
            if not _m:
                _m = re.search(r'--script-name\s+(\S+)', _a) if _a else None
            if _m:
                _ck = _m.group(1)
        if not _ck:
            _ck = script
        ckey = _ck
        # Check if script file exists
        script_file = script
        for prefix in ['node ', 'bash ', 'python3 ']:
            if script.startswith(prefix):
                script_file = script[len(prefix):]
                break
        script_path = os.path.join(GOV_CRAWLER_DIR, script_file) if script_file else ''
        script_exists = os.path.isfile(script_path) if script_path else False
        script_name_key = ''
        script_name_full = ''
        if script:
            script_base = os.path.basename(script).replace('.py', '')
            script_name_full = script_base + '.py'
            if script_base.startswith('crawl_'):
                script_name_key = script_base[6:]
            else:
                script_name_key = script_base
        # Strategy 1 (now primary): match by DB script_name column -> site_name(s) -> db_stats
        # 有唯一标识(ckey, script_name字段/--script-name)的配置: 跳过脚本名聚合,
        # 直接走 site_name fallback 精确匹配(如江都13站各显示各站统计, 不合并)
        stats = {}
        if ckey and ckey != script:
            _keys_to_try = []
        else:
            _keys_to_try = list(dict.fromkeys([k for k in [script_name_full, script_name_key] if k and k in _sn_set]))
        if _keys_to_try:
            try:
                _sites = set()
                for _k in _keys_to_try:
                    _sites |= _sn_to_sites.get(_k, set())
                _total = 0
                _latest = '-'
                _oldest = '-'
                _week_new = 0
                _day_new = 0
                for _site in _sites:
                    _r = db_stats.get(_site)
                    if _r:
                        _total += _r['total']
                        if _r['latest_date'] and _r['latest_date'] > _latest:
                            _latest = _r['latest_date']
                        if _r['oldest_date'] and (_oldest == '-' or _r['oldest_date'] < _oldest):
                            _oldest = _r['oldest_date']
                        _week_new += _r['week_new']
                        _day_new += _r['day_new']
                if _total > 0:
                    stats = {'total': _total, 'latest_date': _latest, 'oldest_date': _oldest, 'week_new': _week_new, 'day_new': _day_new}
            except:
                pass
        if not stats:
            # Fallback: site_name matching (from pre-aggregated db_stats)
            stats = db_stats.get(name, {})
        if not stats:
            # Try fuzzy match by normalized name
            stats = db_stats_norm.get(normalize(name), {})
        if not stats:
            # Strategy: config name contains site_name
            name_norm = normalize(name)
            for sn, v in db_stats_norm.items():
                if sn in name_norm:
                    stats = v
                    break
        if not stats:
            # Strategy: site_name contains config name
            for sn, v in db_stats_norm.items():
                if name_norm in sn:
                    stats = v
                    break
        if not stats:
            # Strategy: shared prefix >= 3 chars
            for sn, v in db_stats_norm.items():
                min_len = min(len(name_norm), len(sn))
                if min_len >= 6:
                    common = 0
                    for i in range(min_len):
                        if name_norm[i] == sn[i]:
                            common += 1
                        else:
                            break
                    if common >= 3:
                        stats = v
                        break
        if not stats:
            # Strategy: base name before separator
            base = name_norm.split('-')[0].split('—')[0].split('·')[0]
            if len(base) >= 2:
                for sn, v in db_stats_norm.items():
                    if base in sn or sn in base:
                        stats = v
                        break
        if not stats:
            # Strategy: strip common suffixes
            common_suffixes = ['通知公告', '公示公告', '环境保护', '环评信息', '环评公告',
                               '环境信息', '政策文件', '新闻专区', '环评审批', '环境影响评价',
                               '高新区动态', '公告公示', '生态环境保护']
            base = name_norm
            for sfx in common_suffixes:
                if base.endswith(sfx):
                    base = base[:-len(sfx)]
                    break
            if len(base) >= 2 and base != name_norm:
                for sn, v in db_stats_norm.items():
                    if base in sn:
                        stats = v
                        break
        if not stats:
            # Strategy: match by script SITE_NAME constant
            site_name_from_script = extract_script_sitename(script_file)
            if site_name_from_script and site_name_from_script in db_stats:
                stats = db_stats[site_name_from_script]
        if not stats:
            # Strategy: match by script filename slug
            if script:
                script_base = os.path.basename(script).replace('.py', '')
                if script_base.startswith('crawl_'):
                    slug = script_base[6:]
                    for sn, v in db_stats.items():
                        if slug and (slug in sn or (len(slug) >= 3 and any(slug[:i] in sn for i in [4,6,8]))):
                            stats = v
                            break
        total = stats.get('total', 0)
        latest = stats.get('latest_date', '-')
        oldest = stats.get('oldest_date', '-')
        week_new = stats.get('week_new', 0)
        day_new = stats.get('day_new', 0)
        active_3d = active_7d = active_30d = active_180d = False
        if latest and latest != '-':
            try:
                from datetime import datetime, timedelta
                ld = datetime.strptime(latest[:10], '%Y-%m-%d')
                delta = datetime.now() - ld
                active_3d = delta <= timedelta(days=3)
                active_7d = delta <= timedelta(days=7)
                active_30d = delta <= timedelta(days=30)
                active_180d = delta <= timedelta(days=180)
            except:
                pass
        # ── 日跑状态: 从 run_logs 查今日状态 ──
        last_run_status = '⏳ 今日待运行'
        last_run_ts = 0
        run_logs_history = []
        try:
            if _rl_load_failed:
                raise RuntimeError('run_logs 预载失败 -> 降级 json')
            # 2026-09-10: 改为查函数级预载的 dict（原先每配置查 3~4 次 sqlite）
            _row = _rl_today.get(ckey) if ckey else None
            if not _row and ckey != script:
                _row = _rl_today.get(script)
            if not _row:
                _row = _rl_today.get(name)
            if _row:
                _st = _row.get('status')
                _nc = _row.get('new_count')
                if _st == '成功':
                    last_run_status = f'✅今日成功 +{_nc}条'
                else:
                    last_run_status = f'❌今日{_st}'
            # 最近5条历史（用于悬停）：多键合并去重后取最近 5
            _seen = set()
            _merged = []
            for _k in dict.fromkeys([ckey, script, name]):
                for _h in (_rl_hist.get(_k) or []):
                    _sig = (_h.get('run_date'), _h.get('status'), _h.get('new_count'))
                    if _sig in _seen:
                        continue
                    _seen.add(_sig)
                    _merged.append(_h)
            _merged.sort(key=lambda _h: (_h.get('run_date') or ''), reverse=True)
            run_logs_history = _merged[:5]
        except:

            try:
                with open('/root/gov_crawler/run_status.json') as _f:
                    _status_map = __import__('json').load(_f)
                status_entry = _status_map.get(script)
                if not status_entry:
                    status_entry = _status_map.get(name)
                if status_entry:
                    last_run_ts = status_entry.get('ts', 0)
                    status_str = status_entry.get('status', '')
                    if status_str:
                        last_run_status = status_str
            except:
                pass
        result.append({
            'name': name, 'script': script, 'group': c.get('group', '未分组'),
            'enabled': c.get('enabled', True), 'note': c.get('note', ''),
            'total': total, 'latest': latest, 'oldest': oldest,
            'week_new': week_new, 'day_new': day_new,
            'active': active_3d, 'active_3d': active_3d, 'active_7d': active_7d,
            'active_30d': active_30d, 'active_180d': active_180d,
            'script_exists': script_exists,
            'last_run_status': last_run_status, 'last_run_ts': last_run_ts,
            'run_logs_history': run_logs_history,
        })
    return result, cfg
def get_analysis_data(period='day', mode='incremental', metric='records', days=90):
    """
    获取分析数据
    period: day/week/month
    mode: incremental / cumulative
    metric: records (数据量) / 脚本数 (脚本数) / both (双指标)
    days: 回溯天数 (0=全部)

    脚本数按唯一脚本文件(.py)统计，脚本的新增日期取该脚本数据
    在 search.db 中最早的 publish_date。
    """
    import json as _json
    import os as _os
    import datetime as _dt
    from collections import defaultdict as _dd

    db = sqlite3.connect(SEARCH_DB)
    db.row_factory = sqlite3.Row
    records_data = []
    spiders_data = []
    labels = []
    cfg_path = "/root/gov_crawler/daily_crawl_config.json"

    # ====== 1. Build script-to-first-date mapping ======
    # Get all unique site_names from DB with their first date
    db_site_dates = {}
    rows = db.execute("""
        SELECT site_name, MIN(date(publish_date)) as d
        FROM gov_raw WHERE publish_date IS NOT NULL AND SUBSTR(publish_date,1,10) != ''
          AND date(publish_date) IS NOT NULL AND date(publish_date) != '1970-01-01'
        GROUP BY site_name
    """).fetchall()
    for r in rows:
        if r['site_name'] and r['d']:
            db_site_dates[r['site_name']] = r['d']
    
    # Read daily_crawl_config.json, match each config entry to DB site_names
    scripts_map = {}  # script_name -> earliest_date (YYYY-MM-DD or None)
    if _os.path.exists(cfg_path):
        try:
            cfg = _json.load(open(cfg_path))
            if isinstance(cfg, list):
                cfg = {"crawlers": cfg}
            crawlers = cfg.get('crawlers', [])
            
            def normalize(s):
                return s.replace(' ', '').replace('—', '-').replace('–', '-') \
                    .replace('（', '(').replace('）', ')').replace('　', '').replace('\xa0', '')
            
            # Index normalized DB names
            db_norm = {normalize(k): k for k in db_site_dates}
            
            for c in crawlers:
                script = c.get('script', '')
                name = c.get('name', '')
                if not script or not name:
                    continue
                
                earliest = None
                name_norm = normalize(name)
                
                # Strategy 0: site_names array
                if 'site_names' in c and isinstance(c['site_names'], list):
                    for sn in c['site_names']:
                        if sn in db_site_dates:
                            d = db_site_dates[sn]
                            if earliest is None or d < earliest:
                                earliest = d
                
                # Strategy 1: exact match
                if earliest is None and name in db_site_dates:
                    earliest = db_site_dates[name]
                
                # Strategy 2: normalized exact match
                if earliest is None and name_norm in db_norm:
                    earliest = db_site_dates[db_norm[name_norm]]
                
                # Strategy 3: strip (#NNN) suffix
                if earliest is None:
                    import re as _re
                    clean = _re.sub(r'\(#\d+\)', '', name).strip()
                    clean_norm = normalize(clean)
                    if clean_norm in db_norm:
                        earliest = db_site_dates[db_norm[clean_norm]]
                
                # Strategy 4: fuzzy - config name in DB site_name
                if earliest is None:
                    for sn_norm, sn in db_norm.items():
                        if name_norm in sn_norm:
                            d = db_site_dates[sn]
                            if earliest is None or d < earliest:
                                earliest = d
                
                # Strategy 5: DB site_name in config name  
                if earliest is None:
                    for sn_norm, sn in db_norm.items():
                        if sn_norm in name_norm:
                            d = db_site_dates[sn]
                            if earliest is None or d < earliest:
                                earliest = d
                
                # Strategy 6: shared prefix >= 3 chars
                if earliest is None:
                    for sn_norm, sn in db_norm.items():
                        min_len = min(len(name_norm), len(sn_norm))
                        if min_len >= 6:
                            common = 0
                            for i in range(min_len):
                                if name_norm[i] == sn_norm[i]:
                                    common += 1
                                else:
                                    break
                            if common >= 3:
                                d = db_site_dates[sn]
                                if earliest is None or d < earliest:
                                    earliest = d
                
                # Strategy 7: base name before -
                if earliest is None:
                    base = name_norm.split('-')[0].split('—')[0]
                    if len(base) >= 2:
                        for sn_norm, sn in db_norm.items():
                            if base in sn_norm:
                                d = db_site_dates[sn]
                                if earliest is None or d < earliest:
                                    earliest = d
                
                # Strategy 8: strip common suffixes
                if earliest is None:
                    common_suffixes = ['通知公告', '公示公告', '环境保护', '环评信息', '环评公告',
                                       '环境信息', '政策文件', '环评审批', '环境影响评价', '公告公示', '行政许可公示']
                    base = name_norm
                    for sfx in common_suffixes:
                        if base.endswith(sfx):
                            base = base[:-len(sfx)]
                            break
                    if len(base) >= 2 and base != name_norm:
                        for sn_norm, sn in db_norm.items():
                            if base in sn_norm or sn_norm in base:
                                d = db_site_dates[sn]
                                if earliest is None or d < earliest:
                                    earliest = d
                
                scripts_map[script] = earliest
                
                if earliest is None:
                    # Fallback: use file mtime
                    try:
                        mtime = _os.path.getmtime('/root/gov_crawler/' + script)
                        scripts_map[script] = _dt.datetime.fromtimestamp(mtime).strftime('%Y-%m-%d')
                    except:
                        pass
        except Exception as e:
            print(f"get_analysis_data config error: {e}")
        # ====== 2. Compute records data from gov_raw ======
    rec_where = "WHERE publish_date IS NOT NULL AND SUBSTR(publish_date,1,10) != '' AND date(publish_date) IS NOT NULL"
    if days > 0:
        rec_where += f" AND date(publish_date) >= date('now', '-{days} days')"

    if period == 'day':
        rec_date_expr = "date(publish_date)"
        rec_label_expr = "date(publish_date)"
    elif period == 'week':
        rec_date_expr = "date(publish_date, 'weekday 1', '-7 days')"
        rec_label_expr = "strftime('%Y-第%W周', publish_date)"
    else:
        rec_date_expr = "strftime('%Y-%m-01', publish_date)"
        rec_label_expr = "strftime('%Y年%m月', publish_date)"

    q_rec = f"""
        SELECT {rec_date_expr} as period_date,
               {rec_label_expr} as period_label,
               COUNT(*) as cnt
        FROM gov_raw
        {rec_where}
        GROUP BY period_date
        ORDER BY period_date ASC
    """
    rec_rows = db.execute(q_rec).fetchall()
    rec_labels = [r['period_label'] for r in rec_rows]
    rec_inc = [r['cnt'] for r in rec_rows]

    # ====== 3. Compute spider data from scripts_map ======
    # Filter scripts by first_date within range
    filtered_scripts = {}
    for s, d in scripts_map.items():
        if d is None:
            # No data match - use file timestamp as proxy for "added date"
            try:
                mtime = _os.path.getmtime(f'/root/gov_crawler/{s}')
                d = _dt.datetime.fromtimestamp(mtime).strftime('%Y-%m-%d')
            except:
                continue
        if days > 0:
            cutoff = (_dt.datetime.now() - _dt.timedelta(days=days)).strftime('%Y-%m-%d')
            if d < cutoff:
                continue
        # Determine period label for this script's date
        if period == 'day':
            plabel = d
        elif period == 'week':
            dd = _dt.datetime.strptime(d, '%Y-%m-%d').date()
            monday = dd - _dt.timedelta(days=dd.weekday())
            plabel = monday.strftime('%Y-第%W周')
        else:
            plabel = d[:7].replace('-', '年') + '月'
        filtered_scripts[s] = (d, plabel)

    # Count scripts per period
    spi_period_cnt = _dd(int)
    for s, (d, plabel) in sorted(filtered_scripts.items(), key=lambda x: x[1][0]):
        spi_period_cnt[plabel] += 1
    spi_labels = sorted(spi_period_cnt.keys())
    spi_inc = [spi_period_cnt[l] for l in spi_labels]

    # ====== 4. Merge records + 脚本数 ======
    all_labels = sorted(set(rec_labels + spi_labels))
    rec_inc_dict = dict(zip(rec_labels, rec_inc))
    spi_inc_dict = dict(zip(spi_labels, spi_inc))

    if mode == 'incremental':
        labels = all_labels
        records_data = [rec_inc_dict.get(l, 0) for l in all_labels]
        spiders_data = [spi_inc_dict.get(l, 0) for l in all_labels]
    else:
        labels = all_labels
        r_carry, s_carry = 0, 0
        for l in all_labels:
            r_carry += rec_inc_dict.get(l, 0)
            s_carry += spi_inc_dict.get(l, 0)
            records_data.append(r_carry)
            spiders_data.append(s_carry)

    db.close()

    result = {
        'labels': labels,
        'record_counts': records_data,
        'spider_counts': spiders_data if spiders_data else None,
        'period': period,
        'mode': mode,
        'metric': metric,
        'days': days,
    }
    return result
def crawler_row_status(c):
    """脚本行状态分类（服务端筛选用；与表格「日跑状态」列同一口径）"""
    run_status = c.get('last_run_status') or ''
    if not c.get('enabled', True):
        return 'disabled'
    if run_status.startswith('✅'):
        return 'success'
    if '超时' in run_status:
        return 'timeout'
    if '不可达' in run_status:
        return 'unreachable'
    if run_status.startswith('❌') or '失败' in run_status or '异常' in run_status or '🐛' in run_status:
        return 'fail'
    return 'pending'


def render_admin_page(page=1, fgroup='all', fstatus='all', fkw='', sort='latest', sdir='desc', per=10,
                      username=''):
    """渲染爬虫管理HTML页面

    2026-09-10: 脚本表改为**服务端分页**（默认每页 10 行）+ 服务端筛选/排序。
    原实现一次性渲染 2086 行 → 2.7MB HTML、页面加载 22s；现在 ≈ 60KB / <4s。
    """
    crawlers, cfg = get_crawler_stats()
    total = len(crawlers)
    import sqlite3; total_records = sqlite3.connect(SEARCH_DB).execute("SELECT COUNT(*) FROM gov_raw").fetchone()[0]; sqlite3.connect(SEARCH_DB).close()
    active_3d = sum(1 for c in crawlers if c['active_3d'])
    active_7d = sum(1 for c in crawlers if c['active_7d'])
    active_30d = sum(1 for c in crawlers if c['active_30d'])
    active_180d = sum(1 for c in crawlers if c['active_180d'])
    enabled_n = sum(1 for c in crawlers if c.get('enabled', True))

    groups = {}
    for c in crawlers:
        g = c['group']
        if g not in groups:
            groups[g] = {'total': 0, 'records': 0, 'active': 0}
        groups[g]['total'] += 1
        groups[g]['records'] += c['total']
        if c['active']:
            groups[g]['active'] += 1

    from datetime import datetime
    now_str = datetime.now().strftime('%Y-%m-%d %H:%M')
    week_total = sum(c['week_new'] for c in crawlers)
    day_total = sum(c['day_new'] for c in crawlers)

    # ── 脚本表：服务端筛选 → 排序 → 分页（2026-09-10）──
    _rows = crawlers
    if fgroup and fgroup != 'all':
        _rows = [c for c in _rows if c['group'] == fgroup]
    if fstatus and fstatus != 'all':
        _rows = [c for c in _rows if crawler_row_status(c) == fstatus]
    if fkw:
        _k = fkw.strip().lower()
        if _k:
            _rows = [c for c in _rows if _k in ((c.get('name') or '') + ' ' + (c.get('script') or '') + ' ' + (c.get('url') or '')).lower()]
    _SORT_KEYS = {
        'status': lambda c: (2 if c['active'] else (1 if c.get('week_new', 0) > 0 else 0)),
        'name':   lambda c: (c.get('name') or ''),
        'total':  lambda c: (c.get('total') or 0),
        'latest': lambda c: ((c.get('latest') or '-')[:10]),
        'week':   lambda c: (c.get('week_new') or 0),
        'active': lambda c: (1 if c['active'] else 0),
        'group':  lambda c: (c.get('group') or ''),
        'script': lambda c: (c.get('script') or ''),
        'run':    lambda c: crawler_row_status(c),
    }
    if sort not in _SORT_KEYS:
        sort = 'latest'
    _sdir = 'asc' if sdir == 'asc' else 'desc'
    try:
        _rows = sorted(_rows, key=_SORT_KEYS[sort], reverse=(_sdir == 'desc'))
    except Exception:
        pass
    filtered_total = len(_rows)
    try:
        per = max(1, int(per))
    except Exception:
        per = 10
    total_pages = max(1, (filtered_total + per - 1) // per)
    try:
        page = int(page)
    except Exception:
        page = 1
    page = max(1, min(page, total_pages))
    page_rows = _rows[(page - 1) * per: page * per]

    def _admin_url(**over):
        """构造保留筛选/排序状态的链接"""
        d = {'g': fgroup, 's': fstatus, 'kw': fkw, 'sort': sort, 'dir': _sdir, 'p': page}
        d.update(over)
        d = {k: v for k, v in d.items() if (v not in ('', 'all', None)) or k == 'g'}
        return '/crawler/admin?' + urllib.parse.urlencode(d)

    def _sort_link(key, label, cls=''):
        """可排序表头（服务端排序，替换原 client-side sortTable）。
        2026-09-11 美化：列宽由内联 style 改为类名；表头链接风格下沉到 .th-sort。"""
        _next = 'desc' if (sort == key and _sdir == 'asc') else 'asc'
        _all = ' '.join([x for x in [('sort-%s' % _sdir) if sort == key else '', cls] if x])
        return ('<th class="%s"><a class="th-sort" href="%s">%s</a></th>'
                % (_all, _admin_url(sort=key, dir=_next, p=1), label))

    def _pager_html():
        """脚本表分页条（复用统一 PAGER_CSS 卡片样式）"""
        if total_pages <= 1:
            return ''
        html = PAGER_CSS + '<div class="pagination">'
        html += '<span class="page-info">第 <b>%d</b> / %d 页 · 本页 %d 条 · 共 %d 个脚本</span>' % (
            page, total_pages, len(page_rows), filtered_total)
        if page > 1:
            html += '<a class="page-btn" href="%s">‹ 上一页</a>' % _admin_url(p=page - 1)
        _pset = {1, total_pages}
        for _i in range(max(1, page - 2), min(total_pages, page + 2) + 1):
            _pset.add(_i)
        _last = 0
        for _p in sorted(_pset):
            if _p - _last > 1:
                html += '<span class="page-ellipsis">…</span>'
            _act = ' active' if _p == page else ''
            html += '<a class="page-btn%s" href="%s">%d</a>' % (_act, _admin_url(p=_p), _p)
            _last = _p
        if page < total_pages:
            html += '<a class="page-btn" href="%s">下一页 ›</a>' % _admin_url(p=page + 1)
        html += '</div>'
        return html

    # 状态筛选链接（服务端筛选，替换原 client-side filterStatus）
    # 2026-09-11 美化：状态色由内联 style 改为 .sf-chip.sf-<状态> 语义类（色值见 CSS）
    _sf_items = [('all', '全部'), ('success', '✅ 成功'), ('timeout', '⏰ 超时'), ('fail', '❌ 失败'),
                 ('unreachable', '🚫 不可达'), ('pending', '⏳ 待运行'), ('disabled', '⏹️ 禁用')]
    _sf_html = ''
    for _key, _label in _sf_items:
        _act = ' active' if _key == (fstatus or 'all') else ''
        _sf_html += ('<a class="sf-chip sf-%s%s" href="%s">%s</a>'
                     % (_key, _act, _admin_url(s=_key, p=1), _label))

    parts = []

    # 加载已保存的排除关键词
    saved_exclude = ""
    try:
        with open(ADMIN_EXCLUDE_PATH) as f:
            saved_exclude = f.read().strip()
    except:
        pass

    parts.append(f'''<!DOCTYPE html>
<html lang="zh-cn">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width, initial-scale=1">
<title>爬虫管理面板</title>
<script src="/static/chart.umd.min.js"></script>
    <script>var savedExcludeKeywords = {json.dumps(saved_exclude)};</script>
<style>
* {{ margin:0; padding:0; box-sizing:border-box; }}
body {{ font-family:-apple-system,BlinkMacSystemFont,"Segoe UI","Microsoft YaHei","PingFang SC",sans-serif; background:#f1f3f5; color:#202124; }}
.wrap {{ max-width:1400px; margin:0 auto; padding:18px 20px 56px; }}
/* ── 顶部标题条 ── */
.page-head {{ display:flex; align-items:center; justify-content:space-between; gap:12px; flex-wrap:wrap; margin-bottom:14px; }}
.ph-left {{ display:flex; align-items:baseline; gap:10px; }}
.ph-logo {{ font-weight:700; color:#1a73e8; font-size:17px; letter-spacing:-.4px; }}
.ph-title {{ font-size:19px; font-weight:600; color:#1a1a2e; }}
/* ── 页签（分段控件）── */
.tabs {{ display:inline-flex; gap:4px; padding:4px; background:#fff; border:1px solid #e9ecef; border-radius:12px; margin-bottom:14px; box-shadow:0 1px 3px rgba(0,0,0,.05); }}
.tab-btn {{ padding:7px 18px; border:none; background:transparent; border-radius:9px; cursor:pointer; font-size:13.5px; font-weight:500; color:#3c4043; font-family:inherit; }}
.tab-btn.active {{ background:#1a73e8; color:#fff; }}
.tab-btn:hover:not(.active) {{ background:#eef4ff; color:#1a73e8; }}
.tab-content {{ display:none; }}
.tab-content.active {{ display:block; }}
/* ── 统计卡 ── */
.stats-row {{ display:flex; gap:12px; flex-wrap:wrap; margin-bottom:12px; }}
.stat-card {{ background:#fff; border:1px solid #e9ecef; border-radius:12px; padding:14px 18px; flex:1; min-width:130px; box-shadow:0 1px 3px rgba(0,0,0,.05); }}
.stat-card .num {{ font-size:26px; font-weight:700; color:#202124; font-variant-numeric:tabular-nums; line-height:1.15; }}
.stat-card .label {{ font-size:12.5px; color:#5f6368; margin-top:5px; }}
.stat-card .label .sub {{ color:#8a9099; }}
.stat-card.green .num {{ color:#1e7d45; }}
.stat-card.blue .num {{ color:#1a73e8; }}
.stat-card.red .num {{ color:#d93025; }}
.active-card {{ flex:2; min-width:250px; }}
.active-rows {{ display:flex; flex-direction:column; gap:7px; margin-top:8px; }}
.active-row {{ display:flex; align-items:center; gap:8px; font-size:12.5px; }}
.active-row .period {{ width:44px; color:#5f6368; text-align:right; flex-shrink:0; }}
.active-row .bar {{ flex:1; height:6px; background:#eef1f4; border-radius:3px; overflow:hidden; }}
.active-row .bar .fill {{ height:100%; border-radius:3px; transition:width .5s; }}
.active-row .count {{ min-width:40px; font-weight:700; text-align:right; flex-shrink:0; font-variant-numeric:tabular-nums; }}
.active-row:nth-child(1) .bar .fill {{ background:#d93025; }} .active-row:nth-child(1) .count {{ color:#d93025; }}
.active-row:nth-child(2) .bar .fill {{ background:#e37400; }} .active-row:nth-child(2) .count {{ color:#e37400; }}
.active-row:nth-child(3) .bar .fill {{ background:#1a73e8; }} .active-row:nth-child(3) .count {{ color:#1a73e8; }}
.active-row:nth-child(4) .bar .fill {{ background:#1e7d45; }} .active-row:nth-child(4) .count {{ color:#1e7d45; }}
/* ── 三块工具区（卡片化）── */
.groups-box,.status-filters,.exclude-row {{ background:#fff; border:1px solid #e9ecef; border-radius:12px; padding:11px 14px; margin-bottom:12px; box-shadow:0 1px 3px rgba(0,0,0,.05); display:flex; gap:7px; flex-wrap:wrap; align-items:center; }}
.groups-box {{ max-height:104px; overflow-y:auto; }}
.hint {{ font-size:12.5px; color:#5f6368; }}
/* ── 胶囊：分组筛选 ── */
.filter-btn {{ padding:5px 12px; border:1px solid #dfe3e8; border-radius:999px; background:#fff; cursor:pointer; font-size:12.5px; color:#3c4043; text-decoration:none; display:inline-block; }}
.filter-btn:hover {{ background:#eef4ff; border-color:#1a73e8; color:#1a73e8; text-decoration:none; }}
.filter-btn.active {{ background:#1a73e8; color:#fff; border-color:#1a73e8; font-weight:600; }}
/* ── 胶囊：状态筛选（按状态着色，原先是内联颜色）── */
.sf-chip {{ padding:4px 12px; border:1px solid #d7dbe0; border-radius:999px; background:#fff; font-size:12px; text-decoration:none; display:inline-block; color:#5f6368; }}
.sf-chip.sf-all {{ border-color:#c8ccd2; color:#3c4043; }}
.sf-chip.sf-all.active {{ background:#3c4043; color:#fff; }}
.sf-chip.sf-success {{ border-color:#1e7d45; color:#1e7d45; }}
.sf-chip.sf-success.active {{ background:#1e7d45; color:#fff; }}
.sf-chip.sf-timeout {{ border-color:#b26a00; color:#b26a00; }}
.sf-chip.sf-timeout.active {{ background:#b26a00; color:#fff; }}
.sf-chip.sf-fail {{ border-color:#d93025; color:#d93025; }}
.sf-chip.sf-fail.active {{ background:#d93025; color:#fff; }}
.sf-chip.sf-unreachable {{ border-color:#7b3fa0; color:#7b3fa0; }}
.sf-chip.sf-unreachable.active {{ background:#7b3fa0; color:#fff; }}
.sf-chip.sf-pending {{ border-color:#8a9099; }}
.sf-chip.sf-pending.active {{ background:#8a9099; color:#fff; }}
.sf-chip.sf-disabled {{ border-color:#c8ccd2; color:#8a9099; }}
.sf-chip.sf-disabled.active {{ background:#8a9099; color:#fff; }}
/* ── 关键词保存 ── */
.exclude-row input {{ padding:6px 12px; border:1px solid #dfe3e8; border-radius:9px; font-size:13px; outline:none; width:260px; font-family:inherit; }}
.exclude-row input:focus {{ border-color:#1a73e8; box-shadow:0 0 0 3px rgba(26,115,232,.12); }}
.btn-save {{ padding:6px 14px; border:1px solid #b26a00; border-radius:999px; background:#fff; color:#b26a00; cursor:pointer; font-size:13px; font-family:inherit; }}
.btn-save:hover {{ background:#fff7ec; }}
.save-status {{ font-size:12.5px; color:#5f6368; }}
.updated {{ font-size:12.5px; color:#8a9099; margin:0 0 12px 2px; }}
/* ── 手动日跑 ── */
.run-card {{ cursor:pointer; background:#f4fbf6; border:1px dashed #1e7d45; text-align:center; }}
.run-card:hover {{ background:#eaf7ee; }}
.run-card .num {{ font-size:22px; color:#1e7d45; }}
.run-log {{ display:none; background:#1e1e1e; color:#7ee787; padding:12px 14px; border-radius:12px; margin:0 0 14px; max-height:300px; overflow-y:auto; font-family:ui-monospace,SFMono-Regular,Menlo,monospace; font-size:11.5px; white-space:pre-wrap; }}
/* ── 表格 ── */
.table-wrap {{ background:#fff; border:1px solid #e9ecef; border-radius:12px; box-shadow:0 1px 3px rgba(0,0,0,.05); overflow:auto; }}
table {{ width:100%; border-collapse:collapse; font-size:13px; }}
th {{ position:sticky; top:0; z-index:2; background:#f8f9fb; color:#3c4043; padding:10px 12px; text-align:left; font-weight:600; font-size:12.5px; white-space:nowrap; border-bottom:1px solid #e9ecef; cursor:pointer; user-select:none; }}
th:hover {{ color:#1a73e8; }}
th a, a.th-sort {{ color:inherit; text-decoration:none; display:block; cursor:pointer; }}
th.sort-asc::after {{ content:" ▲"; font-size:10px; color:#1a73e8; }}
th.sort-desc::after {{ content:" ▼"; font-size:10px; color:#1a73e8; }}
td {{ padding:9px 12px; border-bottom:1px solid #f1f3f5; vertical-align:middle; }}
tbody tr:last-child td {{ border-bottom:none; }}
tbody tr:hover {{ background:#f7faff; }}
tbody tr.row-off {{ opacity:.5; }}
.ta-r {{ text-align:right; font-variant-numeric:tabular-nums; }}
.tc {{ text-align:center; }}
.active-col {{ text-align:center; }}
.ell {{ max-width:220px; overflow:hidden; text-overflow:ellipsis; white-space:nowrap; }}
.ell.mono {{ font-size:11.5px; font-family:ui-monospace,SFMono-Regular,Menlo,monospace; max-width:190px; }}
.nowrap {{ white-space:nowrap; }}
.dash {{ color:#c8ccd2; font-size:11.5px; }}
.url-link {{ color:#1a73e8; font-size:11.5px; text-decoration:none; }}
.url-link:hover {{ text-decoration:underline; }}
.muted {{ color:#8a9099; }}
.missing-script {{ color:#d93025; font-weight:600; }}
.run-state {{ font-size:12px; text-align:center; cursor:default; white-space:nowrap; }}
.run-ok {{ color:#1e7d45; }} .run-warn {{ color:#b26a00; }} .run-block {{ color:#7b3fa0; }} .run-bad {{ color:#d93025; }} .run-idle {{ color:#8a9099; }}
.btn-toggle {{ padding:3px 11px; border-radius:999px; border:1px solid #dddfe3; cursor:pointer; font-size:12px; font-family:inherit; }}
.btn-toggle.on {{ background:#1e7d45; color:#fff; border-color:#1e7d45; }}
.btn-toggle.off {{ background:#f5f6f8; color:#8a9099; }}
.badge {{ display:inline-block; padding:1px 8px; border-radius:999px; font-size:11px; font-weight:500; }}
.badge-green {{ background:#e6f4ea; color:#1e7d45; }}
.badge-red {{ background:#fce8e6; color:#c5221f; }}
.badge-gray {{ background:#f1f3f4; color:#5f6368; }}
.w-num {{ width:64px; }} .w-date {{ width:92px; }} .w-week {{ width:74px; }} .w-act {{ width:60px; }}
.w-run {{ width:96px; }} .w-toggle {{ width:76px; }}
/* ── 分析视图 ── */
.analysis-controls {{ display:flex; gap:12px; flex-wrap:wrap; margin-bottom:12px; align-items:center; background:#fff; border:1px solid #e9ecef; border-radius:12px; padding:11px 14px; box-shadow:0 1px 3px rgba(0,0,0,.05); }}
.control-group {{ display:flex; gap:6px; align-items:center; }}
.control-group label {{ font-size:12.5px; color:#5f6368; }}
.control-group select, .control-group button {{ padding:6px 12px; border:1px solid #dfe3e8; border-radius:9px; background:#fff; font-size:13px; cursor:pointer; font-family:inherit; color:#3c4043; }}
.control-group button.active-ctrl {{ background:#1a73e8; color:#fff; border-color:#1a73e8; }}
.chart-container {{ background:#fff; border:1px solid #e9ecef; border-radius:12px; padding:18px; box-shadow:0 1px 3px rgba(0,0,0,.05); margin-bottom:16px; min-height:350px; }}
.chart-summary {{ display:flex; gap:16px; flex-wrap:wrap; margin-top:12px; }}
.chart-stat {{ font-size:13.5px; color:#3c4043; }}
.chart-stat span {{ font-weight:700; color:#1a73e8; }}
.cs-sep {{ border-right:1px solid #e9ecef; padding-right:16px; }}
.cs-sepl {{ border-left:1px solid #e9ecef; padding-left:16px; color:#5f6368; }}
.cs-b {{ color:#1a73e8; }} .cs-g {{ color:#1e7d45; }} .cs-err {{ color:#d93025; }}
.legend {{ font-size:12.5px; color:#5f6368; margin-bottom:10px; display:flex; gap:14px; flex-wrap:wrap; align-items:center; }}
.legend i {{ display:inline-block; width:12px; height:3px; margin-right:5px; vertical-align:middle; border-radius:2px; }}
.legend i.b {{ background:#1a73e8; }} .legend i.g {{ background:#1e7d45; }}
/* ── 手机端 ── */
@media(max-width:760px) {{
  .wrap {{ padding:12px 12px 44px; }}
  .page-head {{ flex-direction:column; align-items:flex-start; }}
  .tabs {{ width:100%; }}
  .tab-btn {{ flex:1; padding:8px 6px; font-size:12.5px; }}
  .stat-card {{ min-width:calc(50% - 6px); padding:12px 14px; }}
  .stat-card .num {{ font-size:22px; }}
  .active-card {{ min-width:100%; }}
  .groups-box {{ max-height:88px; }}
  .exclude-row input {{ width:100%; }}
  table {{ font-size:12px; }}
  th, td {{ padding:7px 8px; }}
  .ell, .ell.mono {{ max-width:150px; }}
}}
</style>
</head>
<body>
<div class="wrap">
<div class="page-head">
  <div class="ph-left"><span class="ph-logo">BJIIR</span><span class="ph-title">爬虫管理面板</span></div>
  <div class="ph-right">{nav_right_html(username)}</div>
</div>
<div class="tabs">
  <button class="tab-btn active" onclick="switchTab(event,'tab-manage')">📋 管理视图</button>
  <button class="tab-btn" onclick="switchTab(event,'tab-analysis')">📊 分析视图</button>
  <button class="tab-btn" onclick="window.location.href='/'">🔙 返回搜索</button>
</div>

<!-- TAB 1: 管理视图 -->
<div id="tab-manage" class="tab-content active">
<div class="stats-row">
  <div class="stat-card green"><div class="num">{total_records}</div><div class="label">总数据条数</div></div>
  <div class="stat-card blue"><div class="num">{total}</div><div class="label">爬虫数</div></div>
  <div class="stat-card green"><div class="num">{enabled_n}</div><div class="label">启用中</div></div>
  <div class="stat-card active-card">
    <div class="label">站点活跃度 <span class="sub">（有新增的站点数）</span></div>
    <div class="active-rows">
      <div class="active-row"><span class="period">近3日</span><div class="bar"><div class="fill" style="width:{active_3d/total*100 if total else 0:.0f}%"></div></div><span class="count">{active_3d}</span></div>
      <div class="active-row"><span class="period">近7日</span><div class="bar"><div class="fill" style="width:{active_7d/total*100 if total else 0:.0f}%"></div></div><span class="count">{active_7d}</span></div>
      <div class="active-row"><span class="period">近1月</span><div class="bar"><div class="fill" style="width:{active_30d/total*100 if total else 0:.0f}%"></div></div><span class="count">{active_30d}</span></div>
      <div class="active-row"><span class="period">近半年</span><div class="bar"><div class="fill" style="width:{active_180d/total*100 if total else 0:.0f}%"></div></div><span class="count">{active_180d}</span></div>
    </div>
  </div>
  <div class="stat-card green"><div class="num">+{day_total}</div><div class="label">今日新增</div></div>
  <div class="stat-card blue"><div class="num">+{week_total}</div><div class="label">近7日新增</div></div>
  <div class="stat-card run-card" id="daily-run-card" onclick="runDaily()">
    <div class="num" id="daily-run-icon">▶️</div>
    <div class="label" id="daily-run-label">手动日跑</div>
  </div>
</div>
<div class="run-log" id="daily-run-log"></div>
<div class="updated">更新时间: {now_str}</div>
<div class="exclude-row">
  <span class="hint">关键词（逗号或空格分隔）</span>
  <input type="text" id="exclude-keywords" placeholder="关键词（逗号或空格分隔）" oninput="applyExclude()">
  <button class="btn-save" onclick="saveExclude()">💾 保存关键词</button>
  <span id="exclude-status" class="save-status" style="display:none;"></span>
</div>
<div class="filters groups-box">
  <span class="hint">分组</span>
  <a class="filter-btn{' active' if fgroup == 'all' else ''}" href="{_admin_url(g='all', p=1)}">全部 ({total})</a>
''')
    for gname, gdata in sorted(groups.items()):
        _gact = ' active' if fgroup == gname else ''
        parts.append(f'  <a class="filter-btn{_gact}" href="{_admin_url(g=gname, p=1)}">{gname} ({gdata["total"]})</a>\n')
    parts.append('</div>')
    parts.append('<div class="status-filters"><span class="hint">状态筛选</span>' + _sf_html + '</div>')
    parts.append('<div class="table-wrap"><table id="crawler-table"><thead><tr>'
                 + _sort_link('status', '状态')
                 + _sort_link('name', '名称')
                 + _sort_link('total', '条数', 'w-num')
                 + _sort_link('latest', '最新', 'w-date')
                 + _sort_link('week', '7日新增', 'w-week')
                 + _sort_link('active', '活跃', 'w-act')
                 + _sort_link('group', '分组')
                 + _sort_link('script', '脚本')
                 + '<th class="w-run">日跑状态</th><th class="w-toggle">启用</th>'
                 + '</tr></thead><tbody>')
    for c in page_rows:
        active_marker = '🟢' if c['active'] else ('🟡' if c['week_new'] > 0 else '⚪')
        script_display = c['script']
        if not c['script_exists']:
            script_display = f'<span class="missing-script">⚠ {c["script"]}</span>'
        elif not c.get('enabled', True):
            script_display = f'<span class="muted">⏹️ {c["script"]}</span>'
        # 从脚本源码提取真实请求URL（config url字段优先覆盖）
        _display_url = c.get('url', '') or c.get('source_url', '')
        # 2026-09-11: 原「来源URL」列已按用户要求删除（历史残留，10 行里 9 行是 "-"）。
        # _display_url 仍保留 —— 它进 data-search，客户端搜索仍能按 URL 命中。

        run_status = c['last_run_status']
        # 构建历史悬停信息
        history_titles = []
        for h in c.get('run_logs_history', []):
            _d = h['run_date'][:10]
            _s = '✅' if h['status'] == '成功' else '❌'
            _n = f"+{h['new_count']}条" if h['new_count'] else ""
            history_titles.append(f"{_d} {_s}{h['status']} {_n}")
        status_title = ' | '.join(history_titles[:5]) if history_titles else ''
        if run_status.startswith('⏳'):
            # 今日待运行 — 显示最近一次历史
            if history_titles:
                status_title = f"最近: {history_titles[0]}" + (f"\n{history_titles[1]}" if len(history_titles) > 1 else "")
        # 颜色：新格式(✅今日/❌今日/⏳) + 旧格式(✅/⏰/🚫/❌)兼容
        # 2026-09-11 美化：内联 color → 语义类 .run-ok/.run-warn/.run-block/.run-bad/.run-idle
        run_cls = ('run-ok' if run_status.startswith('✅') else
                   'run-warn' if ('⏰' in run_status or '超时' in run_status) else
                   'run-block' if ('🚫' in run_status or '不可达' in run_status) else
                   'run-bad' if (run_status.startswith('❌') or '🐛' in run_status
                                 or '失败' in run_status or '异常' in run_status) else 'run-idle')
        search_text = (c['name'] + ' ' + c['script'] + ' ' + (_display_url or '')).lower().replace('"', '')
        # 状态分类统一走 crawler_row_status（与顶部状态筛选、服务端筛选同一口径）
        _row_status = crawler_row_status(c)
        _tr_extra = ' class="row-off"' if not c.get('enabled', True) else ''
        _enabled = c.get('enabled', True)
        _toggle_btn = ('<button class="btn-toggle %s" onclick="toggleEnabled(&#39;%s&#39;)" title="%s">%s</button>') % (
            'on' if _enabled else 'off',
            str(c['name']).replace("'", "&#39;"),
            '点击禁用' if _enabled else '点击启用',
            '✅ 启用' if _enabled else '⏹️ 禁用',
        )
        parts.append(f'''<tr data-search=\\\"{search_text}\\\" data-status=\\\"{_row_status}\\\"{_tr_extra}>
  <td class="tc">{active_marker}</td>
  <td class="ell" title="{c['name']}">{c['name']}</td>
  <td class="ta-r">{c['total']}</td>
  <td class="nowrap">{(c['latest'] or '')[:10] if (c['latest'] or '') != '-' else '-'}</td>
  <td class="ta-r">{c['week_new']}</td>
  <td class="active-col">{'✅' if c['active'] else '❌'}</td>
  <td>{c['group']}</td>
  <td class="ell mono">{script_display}</td>
  <td class="run-state {run_cls}" title="{status_title}">{run_status}</td>
  <td class="tc">{_toggle_btn}</td>
</tr>
''')
    parts.append('</tbody></table></div>')
    parts.append(_pager_html())
    # 2026-09-11 修复（既有 bug）：原先这里少了 </div>，#tab-manage 一直没闭合 →
    # 浏览器把下面的 #tab-analysis 解析成 #tab-manage 的子节点 →
    # 切到「分析视图」时外层已 display:none，整块（含图表）被吞，页签点了白屏。
    # 实测：analysisIsDescendantOfManage=true、图表画布 0×0。补上即成为 .wrap 的兄弟节点。
    parts.append('</div><!-- /tab-manage -->')
    parts.append('''<div id="tab-analysis" class="tab-content">
  <div class="analysis-controls">
    <div class="control-group">
      <label>时间轴</label>
      <select id="period-select" onchange="loadChart()">
        <option value="day">按日</option>
        <option value="week">按周</option>
        <option value="month">按月</option>
      </select>
    </div>
    <div class="control-group">
      <label>纵轴</label>
      <button id="mode-inc" class="active-ctrl" onclick="setMode('incremental')">单日增量</button>
      <button id="mode-cum" onclick="setMode('cumulative')">累计总量</button>
    </div>
    <div class="control-group">
      <label>指标</label>
      <button id="metric-both" class="active-ctrl" onclick="setMetric('both')">双指标</button>
      <button id="metric-records" onclick="setMetric('records')">数据量</button>
      <button id="metric-spiders" onclick="setMetric('spiders')">脚本数</button>
    </div>
    <div class="control-group">
      <label>时间范围</label>
      <select id="range-select" onchange="loadChart()">
        <option value="30">近30天</option>
        <option value="90" selected>近90天</option>
        <option value="180">近半年</option>
        <option value="365">近一年</option>
        <option value="0">全部</option>
      </select>
    </div>
  </div>
  <div class="legend">
    <span><i class="b"></i>数据量</span>
    <span><i class="g"></i>脚本数</span>
  </div>
  <div class="chart-container">
    <canvas id="analysis-chart" height="80"></canvas>
  </div>
  <div id="chart-summary" class="chart-summary"></div>
</div>  <!-- end tab-analysis -->

<script>
var currentMode = 'incremental';
var currentMetric = 'both';
var chartInstance = null;

function switchTab(event, tabId) {
  document.querySelectorAll('.tab-content').forEach(t => t.classList.remove('active'));
  document.querySelectorAll('.tab-btn').forEach(b => b.classList.remove('active'));
  document.getElementById(tabId).classList.add('active');
  event.target.classList.add('active');
  if (tabId === 'tab-analysis') loadChart();
}

function setMode(mode) {
  currentMode = mode;
  document.getElementById('mode-inc').classList.toggle('active-ctrl', mode === 'incremental');
  document.getElementById('mode-cum').classList.toggle('active-ctrl', mode === 'cumulative');
  loadChart();
}

function setMetric(metric) {
  currentMetric = metric;
  document.getElementById('metric-both').classList.toggle('active-ctrl', metric === 'both');
  document.getElementById('metric-records').classList.toggle('active-ctrl', metric === 'records');
  document.getElementById('metric-spiders').classList.toggle('active-ctrl', metric === 'spiders');
  loadChart();
}


function saveExclude() {
  const kw = document.getElementById('exclude-keywords').value.trim();
  const status = document.getElementById('exclude-status');
  status.style.display = 'inline';
  status.textContent = '⏳ 保存中...';
  fetch('/crawler/admin/save-exclude?kw=' + encodeURIComponent(kw))
    .then(r => r.json())
    .then(data => {
      if (data.success) {
        // 保存成功 → 短暂提示后刷新页面（刷新后输入框自动回填已保存关键词）
        status.style.color = '#1e7d45';
        status.style.fontWeight = 'bold';
        status.style.background = '#eaf7ee';
        status.style.padding = '4px 12px';
        status.style.borderRadius = '12px';
        status.style.border = '1px solid #1e7d45';
        status.textContent = kw ? ('✅ 已保存关键词：' + kw + '，即将刷新') : '✅ 已清除关键词，即将刷新';
        setTimeout(() => { location.reload(); }, 800);
      } else {
        status.style.color = '#d93025';
        status.style.fontWeight = 'bold';
        status.style.background = '#fce8e6';
        status.style.padding = '4px 12px';
        status.style.borderRadius = '12px';
        status.style.border = '1px solid #d93025';
        status.textContent = '❌ 失败: ' + (data.error || '');
      }
      setTimeout(() => { status.style.display = 'none'; status.style.color = '#8a9099'; status.style.fontWeight = 'normal'; status.style.background = ''; status.style.padding = ''; status.style.borderRadius = ''; status.style.border = ''; }, 5000);
    })
    .catch(e => {
      status.style.color = '#d93025';
      status.style.fontWeight = 'bold';
      status.style.background = '#fce8e6';
      status.style.padding = '4px 12px';
      status.style.borderRadius = '12px';
      status.style.border = '1px solid #d93025';
      status.textContent = '❌ 请求失败';
      setTimeout(() => { status.style.display = 'none'; status.style.color = '#8a9099'; status.style.fontWeight = 'normal'; status.style.background = ''; status.style.padding = ''; status.style.borderRadius = ''; status.style.border = ''; }, 5000);
    });
}

function applyExclude() {
  // 仅用于保存关键词（白名单），实际过滤在 /crawler/ 搜索页生效（handle_db_list 服务端过滤）
  // 语义：不再排除，而是"仅限给定关键词"（OR 匹配任一即保留）
}

// 页面加载时填充已保存的排除关键词
document.addEventListener('DOMContentLoaded', function() {
  var input = document.getElementById('exclude-keywords');
  if (input && typeof savedExcludeKeywords !== 'undefined') {
    input.value = savedExcludeKeywords;
  }
});

function loadChart() {
  var period = document.getElementById('period-select').value;
  var days = document.getElementById('range-select').value;
  var mode = currentMode;
  var metric = currentMetric;
  var summaryDiv = document.getElementById('chart-summary');
  summaryDiv.innerHTML = '<div class="chart-stat">⏳ 加载中...</div>';
  fetch('/crawler/admin/analysis?period=' + period + '&mode=' + mode + '&metric=' + metric + '&days=' + days)
    .then(r => r.json())
    .then(data => {
      if (!data.labels || data.labels.length === 0) {
        summaryDiv.innerHTML = '<div class="chart-stat">暂无数据</div>';
        return;
      }
      renderChart(data);
      var unit = period === 'month' ? '月' : (period === 'week' ? '周' : '日');
      var rc = data.record_counts || [];
      var sc = data.spider_counts || [];
      // Records stats
      var r_total = rc.length > 0 ? rc[rc.length - 1] : 0;
      var r_sum = rc.reduce(function(a,b){return a+b}, 0);
      var r_avg = rc.length > 0 ? (r_sum / rc.length).toFixed(1) : '0';
      var r_max = rc.length > 0 ? Math.max.apply(null, rc) : 0;
      // Spider stats
      var s_total = sc.length > 0 ? sc[sc.length - 1] : 0;
      var s_sum = sc.reduce(function(a,b){return a+b}, 0);
      var s_avg = sc.length > 0 ? (s_sum / sc.length).toFixed(1) : '0';
      var s_max = sc.length > 0 ? Math.max.apply(null, sc) : 0;
      summaryDiv.innerHTML =
        '<div class="chart-stat cs-sep">' +
          '<b class="cs-b">📊 数据量</b> 总计: <span>' + r_total + '</span> | ' +
          '均值: <span>' + r_avg + '</span>/' + unit + ' | 峰值: <span>' + r_max + '</span>/' + unit +
        '</div>' +
        '<div class="chart-stat">' +
          '<b class="cs-g">📜 脚本数</b> 总计: <span>' + s_total + '</span> | ' +
          '均值: <span>' + s_avg + '</span>/' + unit + ' | 峰值: <span>' + s_max + '</span>/' + unit +
        '</div>' +
        '<div class="chart-stat cs-sepl">🗓 <span>' + data.labels.length + '</span> 个时间区间</div>';
    })
    .catch(function(err) {
      summaryDiv.innerHTML = '<div class="chart-stat cs-err">加载失败: ' + err.message + '</div>';
    });
}

function renderChart(data) {
  var ctx = document.getElementById('analysis-chart').getContext('2d');
  if (chartInstance) chartInstance.destroy();

  var isCum = currentMode === 'cumulative';
  var labelSuffix = isCum ? '累计' : '增量';

  var labels = data.labels;
  var rc = data.record_counts || [];
  var sc = data.spider_counts || [];

  if (labels.length > 120) {
    var step = Math.ceil(labels.length / 120);
    var nl = [], nrc = [], nsc = [];
    for (var i = 0; i < labels.length; i += step) {
      nl.push(labels[i]);
      nrc.push(rc[i] || 0);
      nsc.push(sc[i] || 0);
    }
    if (nl[nl.length-1] !== labels[labels.length-1]) {
      nl.push(labels[labels.length-1]);
      nrc.push(rc[rc.length-1] || 0);
      nsc.push(sc[sc.length-1] || 0);
    }
    labels = nl; rc = nrc; sc = nsc;
  }

  var datasets = [];
  var activeMetric = currentMetric;

  if (activeMetric === 'records' || activeMetric === 'both') {
    datasets.push({
      label: '数据量 (' + labelSuffix + ')', data: rc, fill: true,
      backgroundColor: 'rgba(26,115,232,0.15)',
      borderColor: 'rgba(26,115,232,0.85)',
      pointBackgroundColor: 'rgba(52,152,219,1)',
      pointRadius: 2, pointHoverRadius: 5,
      borderWidth: 2, tension: 0.35,
    });
  }

  if (activeMetric === 'spiders' || activeMetric === 'both') {
    datasets.push({
      label: '脚本数 (' + labelSuffix + ')', data: sc, fill: true,
      backgroundColor: 'rgba(30,125,69,0.12)',
      borderColor: 'rgba(30,125,69,0.85)',
      pointBackgroundColor: 'rgba(39,174,96,1)',
      pointRadius: 2, pointHoverRadius: 5,
      borderWidth: 2, tension: 0.35,
      yAxisID: 'y',
    });
  }

  chartInstance = new Chart(ctx, {
    type: 'line',
    data: { labels: labels, datasets: datasets },
    options: {
      responsive: true, maintainAspectRatio: false,
      interaction: { mode: 'index', intersect: false },
      plugins: {
        legend: { display: datasets.length > 1, position: 'top' },
        tooltip: { callbacks: { label: function(ctx) { return ctx.dataset.label + ': ' + ctx.parsed.y; } } }
      },
      scales: {
        x: { grid: { display: false }, ticks: { font: { size: 11 }, maxRotation: 45 } },
        y: { beginAtZero: true, grid: { color: 'rgba(0,0,0,0.06)' }, ticks: { font: { size: 11 } } }
      }
    }
  });
}

// Daily run
var dailyCheckTimer = null;
function runDaily() {
  var card = document.getElementById('daily-run-card');
  var icon = document.getElementById('daily-run-icon');
  var label = document.getElementById('daily-run-label');
  var logDiv = document.getElementById('daily-run-log');
  icon.textContent = '⏳'; label.textContent = '启动中...';
  card.style.cursor = 'default'; card.onclick = null;
  fetch('/crawler/admin/run-daily').then(function(r){return r.json()}).then(function(data){
    if (data.running) {
      icon.textContent = '🔄'; label.textContent = '运行中...';
      logDiv.style.display = 'block';
      logDiv.textContent = data.message + '\\n\\n日志即将刷新...';
      dailyCheckTimer = setInterval(checkStatus, 5000);
      checkStatus();
    } else {
      icon.textContent = '❌'; label.textContent = data.message;
      setTimeout(resetButton, 3000);
    }
  }).catch(function(e){ icon.textContent = '❌'; label.textContent = '网络错误'; setTimeout(resetButton, 3000); });
}
function checkStatus() {
  fetch('/crawler/admin/run-status').then(function(r){return r.json()}).then(function(data){
    var icon = document.getElementById('daily-run-icon');
    var label = document.getElementById('daily-run-label');
    var logDiv = document.getElementById('daily-run-log');
    if (data.running && data.log_tail) logDiv.textContent = data.log_tail;
    if (!data.running) {
      icon.textContent = '✅'; label.textContent = '日跑完成';
      if (dailyCheckTimer) { clearInterval(dailyCheckTimer); dailyCheckTimer = null; }
      setTimeout(function(){ location.reload(); }, 3000);
    }
  });
}
/* 2026-09-11: 原 editUrl()（弹 prompt 改「来源URL」）已移入 Archive —— flat-list 时代孤儿，
   UI 无入口调用；配套端点 /crawler/admin/set-url 与 set_crawler_url() 一并移除。 */
function toggleEnabled(name) {
  if (!confirm('确认切换爬虫启用状态?\\n' + name)) return;
  fetch('/crawler/admin/toggle-enabled?name=' + encodeURIComponent(name))
    .then(function(r){return r.json()})
    .then(function(data){
      if (data.success) location.reload();
      else alert('切换失败: ' + (data.error || '未知错误'));
    })
    .catch(function(e){ alert('网络错误: ' + e.message); });
}
function resetButton() {
  var card = document.getElementById('daily-run-card');
  var icon = document.getElementById('daily-run-icon');
  var label = document.getElementById('daily-run-label');
  icon.textContent = '▶️'; label.textContent = '手动日跑';
  card.style.cursor = 'pointer'; card.onclick = runDaily;
}


</script>
</div><!-- /wrap -->
</body></html>''')
    return ''.join(parts)


# 2026-09-11: set_crawler_url(name, url) 已移入 Archive/seturl_orphan_20260911.py
# （flat-list 时代的「改来源URL」入口，UI 无调用方，端点 /crawler/admin/set-url 已删）

def toggle_crawler_enabled(name):
    """切换爬虫配置的启用/禁用状态 (保持原 flat-list 格式)"""
    cfg = json.load(open(CONFIG_PATH))
    is_list = isinstance(cfg, list)
    crawlers = cfg if is_list else cfg.get('crawlers', [])
    for c in crawlers:
        cname = c.get('name', c.get('site_name', ''))
        if cname == name:
            cur = c.get('enabled', True)
            c['enabled'] = not cur
            out = crawlers if is_list else cfg
            json.dump(out, open(CONFIG_PATH, 'w'), ensure_ascii=False, indent=2)
            return {'success': True, 'name': name, 'enabled': c['enabled']}
    return {'success': False, 'error': f'Crawler "{name}" not found'}

def run_single_script(name):
    """运行单个爬虫脚本"""
    import subprocess, json, os, time
    cfg = json.load(open(CONFIG_PATH))
    if isinstance(cfg, list):
        cfg = {"crawlers": cfg}
    script = ''
    for c in cfg['crawlers']:
        if c.get('name', c.get('site_name', '')) == name:
            script = c.get('script', c.get('command', ''))
            break
    if not script:
        return {'success': False, 'error': 'Crawler "%s" not found' % name}
    # 构建命令
    if script.endswith('.sh'):
        cmd = ['bash', os.path.join(GOV_CRAWLER_DIR, script)]
    elif script.endswith('.js'):
        cmd = ['node', os.path.join(GOV_CRAWLER_DIR, script)]
    else:
        cmd = ['python3', os.path.join(GOV_CRAWLER_DIR, script)]
        # 如果有 args，追加
        for c in cfg['crawlers']:
            if c.get('name', '') == name:
                args = c.get('args', '')
                if args:
                    if isinstance(args, list):
                        cmd.extend([str(a) for a in args])
                    else:
                        cmd.extend(str(args).split())
                break
    try:
        result = subprocess.run(cmd, capture_output=True, text=True, timeout=300)
        rc = result.returncode
        output = result.stdout[-500:] + result.stderr[-500:]
        # 更新日跑状态
        now_ts = int(time.time())
        run_record = {
            'script': script,
            'ts': now_ts,
            'rc': rc,
            'has_error': rc != 0,
            'timeout': False,
            'new_count': 0,
            'name': name,
        }
        status_file = '/tmp/crawl_results.json'
        try:
            existing = json.load(open(status_file))
        except:
            existing = []
        existing.append(run_record)
        if len(existing) > 500:
            existing = existing[-500:]
        json.dump(existing, open(status_file, 'w'))
        # 也写入 run_status.json (兼容旧版)
        run_status_file = '/root/gov_crawler/run_status.json'
        run_emoji = '✅' if rc == 0 else '❌'
        try:
            rs = json.load(open(run_status_file))
        except:
            rs = {}
        # Extract new_count from output if possible
        new_count = 0
        for line in output.split('\n'):
            if '新增' in line or 'new' in line:
                import re
                m = re.search(r'(\d+)', line)
                if m:
                    new_count = int(m.group(1))
        rs[name] = {'status': '✅ +%d条' % new_count if rc == 0 else '❌ rc=%d' % rc, 'ts': now_ts}
        json.dump(rs, open(run_status_file, 'w'), ensure_ascii=False)
        # 写入 run_logs DB
        try:
            _db = __import__('sqlite3').connect('/root/search.db')
            _today = __import__('datetime').datetime.now().strftime('%Y-%m-%d')
            _now = __import__('datetime').datetime.now().strftime('%H:%M:%S')
            _status = '成功' if rc == 0 else '失败'
            _script_name = os.path.basename(script) if script else name
            _db.execute(
                "INSERT OR REPLACE INTO run_logs "
                "(run_date, run_time, script_name, config_name, status, new_count, elapsed_seconds, error_detail) "
                "VALUES (?, ?, ?, ?, ?, ?, ?, ?)",
                (_today, _now, _script_name, name, _status, new_count, 0,
                 '' if rc == 0 else f'rc={rc}'))
            _db.commit()
            _db.close()
        except:
            pass
        return {
            'success': rc == 0,
            'name': name,
            'script': script,
            'rc': rc,
            'output': output[:200],
            'message': '已完成 (exit 0)' if rc == 0 else 'rc=%d' % rc
        }
    except subprocess.TimeoutExpired:
        return {'success': False, 'error': '执行超时(300s)', 'name': name}
    except Exception as e:
        return {'success': False, 'error': str(e), 'name': name}


# ─── 手动日跑 ───

RUN_PID_FILE = "/tmp/daily_crawl_run.pid"
RUN_LOG_TAIL = 20

def run_daily_crawl():
    import os, subprocess
    status = get_daily_run_status()
    if status.get('running'):
        return {'success': False, 'message': '日跑已在运行中', 'running': True}
    try:
        pid = os.fork()
        if pid == 0:
            os.setsid()
            subprocess.run(['bash', '/root/daily_incremental.sh'],
                         stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
            try: os.remove(RUN_PID_FILE)
            except: pass
            os._exit(0)
        else:
            with open(RUN_PID_FILE, 'w') as f: f.write(str(pid))
            return {'success': True, 'message': '日跑已启动', 'running': True, 'pid': pid}
    except Exception as e:
        return {'success': False, 'message': f'启动失败: {e}', 'running': False}

def get_daily_run_status():
    import os, subprocess
    if os.path.exists(RUN_PID_FILE):
        try:
            with open(RUN_PID_FILE) as f: pid = int(f.read().strip())
            os.kill(pid, 0)
            log_tail = ''
            try:
                log_tail = subprocess.check_output(
                    ['tail', f'-{RUN_LOG_TAIL}', '/var/log/crawler_daily.log'],
                    timeout=3, stderr=subprocess.DEVNULL).decode('utf-8', errors='replace')
            except: pass
            return {'running': True, 'pid': pid, 'log_tail': log_tail}
        except (OSError, ValueError):
            try: os.remove(RUN_PID_FILE)
            except: pass
    return {'running': False}


if __name__ == '__main__':

    http.server.ThreadingHTTPServer(('0.0.0.0', PORT), SearchHandler).serve_forever()
