drowzeys's picture
v1.0 alpha: keys-Auto Receipts Studio (iPhone / may add Autonomous Lamp Skill)
2edb151 verified
Raw
History Blame Contribute Delete
1.36 kB
#!/usr/bin/env bash
# Gemma 4 12B Unified. GPU util 0.15 (~18.3GiB of 121.7GiB). Never above 0.85.
# BF16 weights ~23GB cannot fit; FP8 ~12.5GB + KV in the rest.
set -euo pipefail
MODEL="${RECEIPT_GEMMA_PATH:-$HOME/models-gemma4-12b-it}"
NAME="${RECEIPT_LLM_MODEL:-google/gemma-4-12B-it}"
HOST="${RECEIPT_VLLM_HOST:-0.0.0.0}"
PORT="${RECEIPT_VLLM_PORT:-8080}"
# User-set 0.15. Fleet hard cap 0.85.
UTIL="${RECEIPT_GPU_MEMORY_UTILIZATION:-0.15}"
MAX_LEN="${RECEIPT_VLLM_MAX_MODEL_LEN:-8192}"
if [[ ! -f "$MODEL/config.json" ]]; then
echo "Gemma checkpoint not found: $MODEL" >&2
exit 1
fi
python3 - "$UTIL" <<'PY'
import sys
util = float(sys.argv[1])
if util > 0.85:
raise SystemExit(f"gpu_memory_utilization {util} > 0.85 hard cap")
print(f"util={util:.4f} pool~{util*121.69:.1f}GiB of 121.7GiB")
print("context: max-model-len default 8192 (receipts). KV estimate at 0.15:")
print(" conservative (48-layer full attn fp16): ~12k tokens")
print(" hybrid (8 full + 40 sliding-1024): ~65k tokens")
print(" model native max_position_embeddings: 262144 (not reachable at 0.15)")
PY
exec vllm serve "$MODEL" \
--served-model-name "$NAME" \
--host "$HOST" \
--port "$PORT" \
--gpu-memory-utilization "$UTIL" \
--max-model-len "$MAX_LEN" \
--max-num-seqs 2 \
--max-num-batched-tokens 2048 \
--quantization fp8 \
--enforce-eager