#!/bin/bash
# 在 GPU 机上部署 Bonsai 2 27B (PQ2_0) 为 systemd 服务，并对齐现有 switch-llm.sh 惯例
set -e
SUDO_PW=yueling408
BIN_DIR=/home/zyw/llama.cpp-bonsai
MODEL_DIR=/home/zyw/models/ternary-bonsai-2-27b

echo "=== [1/6] 目录归位 ==="
mkdir -p "$BIN_DIR" "$MODEL_DIR"
if [ -d /home/zyw/bonsai2/bin ] && [ ! -d "$BIN_DIR/bin" ]; then
  mv /home/zyw/bonsai2/bin "$BIN_DIR/bin"
fi
shopt -s nullglob
for f in /home/zyw/bonsai2/models/*.gguf; do
  b=$(basename "$f")
  # 语义化改名：与官方文件名保持一致
  case "$b" in
    PTQ1_0.gguf)          t=Ternary-Bonsai-2-27B-PTQ1_0.gguf ;;
    PQ2_0.gguf)           t=Ternary-Bonsai-2-27B-PQ2_0.gguf ;;
    mmproj-Q8_0.gguf)     t=Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf ;;
    *)                    t=$b ;;
  esac
  [ -f "$MODEL_DIR/$t" ] || mv "$f" "$MODEL_DIR/$t"
done
chmod +x "$BIN_DIR"/bin/* 2>/dev/null || true
ls -la "$MODEL_DIR"
echo "二进制: $BIN_DIR/bin/llama-server"

echo "=== [2/6] 写 systemd 单元 ==="
cat > /tmp/llama-bonsai2.service <<'EOF'
# /etc/systemd/system/llama-bonsai2.service
# 部署日期: 2026-09-19
# PrismML Bonsai 2 27B —— PQ2_0 三值 packing (2.13 bpw)，需 PrismML fork 的 llama.cpp
# 底座与 Ridge / KO-Ridge 相同（Qwen3.8-27B），仅权重表示不同（三值 + Hadamard 旋转基）
# 与其它 LLM 服务互斥（都占 8080）
# 实测（RTX 5060 Ti 16GB）：tg128 49.7 t/s / pp512 1001 t/s / 262K+Q4_0 KV 显存 13802 MiB
[Unit]
Description=llama.cpp server - Bonsai 2 27B ternary PQ2_0 (262K ctx, vision on GPU)
After=network.target

[Service]
Type=simple
User=zyw
Group=zyw
WorkingDirectory=/home/zyw/models/ternary-bonsai-2-27b
# fork 二进制不自带 CUDA 运行时，需系统 CUDA 13 的 libcudart.so.13 / libcublas.so.13
Environment=LD_LIBRARY_PATH=/usr/local/cuda-13.1/lib64
ExecStart=/home/zyw/llama.cpp-bonsai/bin/llama-server \
  -m /home/zyw/models/ternary-bonsai-2-27b/Ternary-Bonsai-2-27B-PQ2_0.gguf \
  --mmproj /home/zyw/models/ternary-bonsai-2-27b/Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf \
  -ngl 99 -fa on -c 262144 -ctk q4_0 -ctv q4_0 -np 1 \
  --jinja --metrics --warmup --no-context-shift \
  --temp 1.0 --top-p 0.95 --top-k 20 --repeat-penalty 1.0 \
  --threads 8 --threads-batch 16 \
  --host 0.0.0.0 --port 8080 --alias bonsai2-27b
Restart=no
TimeoutStartSec=900
TimeoutStopSec=30
StandardOutput=journal
StandardError=journal
SyslogIdentifier=llama-bonsai2

[Install]
WantedBy=multi-user.target
EOF
echo "$SUDO_PW" | sudo -S mv /tmp/llama-bonsai2.service /etc/systemd/system/llama-bonsai2.service
echo "$SUDO_PW" | sudo -S chown root:root /etc/systemd/system/llama-bonsai2.service
echo "$SUDO_PW" | sudo -S chmod 644 /etc/systemd/system/llama-bonsai2.service
echo "$SUDO_PW" | sudo -S systemctl daemon-reload
# 与其它服务一致：不自启
echo "$SUDO_PW" | sudo -S systemctl disable llama-bonsai2 >/dev/null 2>&1 || true
echo "单元: $(systemctl is-enabled llama-bonsai2 2>&1) / active=$(systemctl is-active llama-bonsai2)"

echo "=== [3/6] 改 switch-llm.sh ==="
python3 - <<'PYEOF'
import re
p="/home/zyw/switch-llm.sh"
s=open(p,encoding="utf-8").read()
orig=s
if "llama-bonsai2" not in s:
    s=s.replace('ALL_UNITS="llama-ridge llama-ridge-long llama-ko-ridge llama-ko-ridge-fast"',
                'ALL_UNITS="llama-ridge llama-ridge-long llama-ko-ridge llama-ko-ridge-fast llama-bonsai2"')
    s=s.replace('  ko-fast)               UNIT=llama-ko-ridge-fast ;;',
                '  ko-fast)               UNIT=llama-ko-ridge-fast ;;\n  bonsai2|bonsai)        UNIT=llama-bonsai2 ;;')
    s=s.replace('*) echo "用法: $0 [ridge-long|ridge-fast|ko-long|ko-fast]"; exit 1 ;;',
                '*) echo "用法: $0 [ridge-long|ridge-fast|ko-long|ko-fast|bonsai2]"; exit 1 ;;')
    s=s.replace('#   ko-fast                            无审核  64K + 原生MTP + 视觉在 GPU     → llama-ko-ridge-fast',
                '#   ko-fast                            无审核  64K + 原生MTP + 视觉在 GPU     → llama-ko-ridge-fast\n'
                '#   bonsai2 (别名 bonsai)               Bonsai 2 27B 三值 PQ2_0 262K + 视觉在 GPU → llama-bonsai2')
    s=s.replace('# 统一 LLM 启停入口：Qwen3.8-27B Ridge(官方) / KO-Ridge(无审核)',
                '# 统一 LLM 启停入口：Qwen3.8-27B Ridge(官方) / KO-Ridge(无审核) / Bonsai 2(三值)')
    open(p,"w",encoding="utf-8").write(s)
    print("switch-llm.sh: 已更新" if s!=orig else "switch-llm.sh: 未变更(模式不匹配)")
else:
    print("switch-llm.sh: 已含 llama-bonsai2，跳过")
PYEOF

echo "=== [4/6] 改 stop-ridge.sh ==="
python3 - <<'PYEOF'
p="/home/zyw/stop-ridge.sh"
s=open(p,encoding="utf-8").read()
if "llama-bonsai2" not in s:
    s=s.replace('for u in llama-ridge llama-ridge-long llama-ko-ridge llama-ko-ridge-fast; do',
                'for u in llama-ridge llama-ridge-long llama-ko-ridge llama-ko-ridge-fast llama-bonsai2; do')
    s=s.replace('# 停止所有 LLM 服务（Ridge 官方版 + KO 无审核版），释放显存',
                '# 停止所有 LLM 服务（Ridge 官方版 + KO 无审核版 + Bonsai 2 三值版），释放显存')
    open(p,"w",encoding="utf-8").write(s); print("stop-ridge.sh: 已更新")
else:
    print("stop-ridge.sh: 已含，跳过")
PYEOF

echo "=== [5/6] 改 ridge-status.sh ==="
python3 - <<'PYEOF'
p="/home/zyw/ridge-status.sh"
s=open(p,encoding="utf-8").read()
if "llama-bonsai2" not in s:
    s=s.replace('echo "=== 单元状态（4 个）==="','echo "=== 单元状态（5 个）==="')
    s=s.replace('for u in llama-ridge llama-ridge-long llama-ko-ridge llama-ko-ridge-fast; do',
                'for u in llama-ridge llama-ridge-long llama-ko-ridge llama-ko-ridge-fast llama-bonsai2; do')
    open(p,"w",encoding="utf-8").write(s); print("ridge-status.sh: 已更新")
else:
    print("ridge-status.sh: 已含，跳过")
PYEOF

echo "=== [6/6] 语法自检 ==="
bash -n /home/zyw/switch-llm.sh && echo "switch-llm.sh 语法 OK"
bash -n /home/zyw/stop-ridge.sh && echo "stop-ridge.sh 语法 OK"
bash -n /home/zyw/ridge-status.sh && echo "ridge-status.sh 语法 OK"
echo "=== 部署完成 ==="
