#!/usr/bin/env bash
# README benchmark: utils/e2e_benchmark.py on BitNet-b1.58-2B-4T (i2_s), 512-token prompt, 128 generated
# tokens, 8 threads, CPU only. Built beforehand with setup_env.py. Metrics are parsed from llama-bench's table.
set -o pipefail
M=/var/tmp/kv-repro/031-microsoft-BitNet/data/BitNet-b1.58-2B-4T/ggml-model-i2_s.gguf
echo "KVERITAS_ARTIFACT role=model name=BitNet-b1.58-2B-4T-i2_s path=../data/BitNet-b1.58-2B-4T/ggml-model-i2_s.gguf visibility=public"
echo "KVERITAS_PHASE name=benchmark"
# e2e_benchmark.py exits 1 even on success (sys.exit(1) is outside its except block), so success is the
# presence of both result rows.
python utils/e2e_benchmark.py -m "$M" -n 128 -p 512 -t 8 2>&1 | tee ../bench.out
grep -q "tg128" ../bench.out && grep -q "pp512" ../bench.out || exit 1
python - ../bench.out <<'PY'
import re, sys
for test, ts, sd in re.findall(r"\|\s*(pp512|tg128)\s*\|\s*([0-9.]+) ± ([0-9.]+)", open(sys.argv[1]).read()):
    print(f"KVERITAS_METRIC name={test}_tokens_per_second value={ts}")
    print(f"KVERITAS_METRIC name={test}_stddev value={sd}")
    print(f"KVERITAS_CLAIM metric={test}_tokens_per_second value={ts}")
PY
