#!/usr/bin/env bash
# README benchmark with the installed vLLM 0.31.0 release (this tree is the v0.31.0 tag): offline
# throughput of Qwen2.5-1.5B-Instruct on 2000 random prompts, 512 input and 256 output tokens.
set -o pipefail
ROW=/var/tmp/kv-repro/027-vllm-project-vllm
# FlashInfer compiles kernels under $HOME at startup; keep that cache inside the row.
export HOME=$ROW/data/home CUDA_HOME=$ROW/env LIBRARY_PATH=$ROW/env/lib
# Run outside this partial tree so the installed package, not ./vllm, is imported.
cd "$ROW/data" || exit 1
echo "KVERITAS_PHASE name=benchmark"
vllm bench throughput --model "$ROW/data/Qwen2.5-1.5B-Instruct" --dataset-name random \
  --input-len 512 --output-len 256 --num-prompts 2000 --output-json "$ROW/bench.json" 2>&1 | tail -5 || exit 1
python - "$ROW/bench.json" <<'PY'
import json, sys
r = json.load(open(sys.argv[1]))
print(f"KVERITAS_METRIC name=elapsed_seconds value={r['elapsed_time']:.6g}")
print(f"KVERITAS_METRIC name=num_requests value={r['num_requests']}")
print(f"KVERITAS_CLAIM metric=requests_per_second value={r['requests_per_second']:.6g}")
print(f"KVERITAS_CLAIM metric=tokens_per_second value={r['tokens_per_second']:.6g}")
PY
