#!/usr/bin/env bash
# README benchmarks: fused linear cross entropy, RMSNorm and SwiGLU, Liger vs Hugging Face, speed and
# memory (baseline: Hugging Face, or plain PyTorch for fused linear cross entropy). Each script appends to benchmark/data/all_benchmark_data.csv; only rows from this GPU are read.
set -o pipefail
ROW=/var/tmp/kv-repro/035-linkedin-Liger-Kernel
export HOME=$ROW/data/home
for b in fused_linear_cross_entropy rms_norm swiglu; do
  echo "KVERITAS_PHASE name=$b"
  python benchmark/scripts/benchmark_$b.py > "$ROW/$b.out" 2>&1 || { tail -20 "$ROW/$b.out"; exit 1; }
done
python - <<'PY'
import csv, statistics
rows = [r for r in csv.DictReader(open("benchmark/data/all_benchmark_data.csv")) if "5060" in r["gpu_name"]]
out = {}
for r in rows:
    key = (r["kernel_name"], r["kernel_operation_mode"], r["metric_name"], r["x_value"], r["extra_benchmark_config_str"])
    out.setdefault(key, {})[r["kernel_provider"]] = float(r["y_value_50"])
ratios = {}
for (k, mode, metric, x, _), v in out.items():
    base = v.get("huggingface", v.get("torch"))
    if "liger" in v and base and v["liger"] > 0:
        ratios.setdefault((k, mode, metric), []).append(base / v["liger"])
for (k, mode, metric), vals in sorted(ratios.items()):
    name = f"{k}_{mode}_{metric}_baseline_over_liger"
    g = statistics.geometric_mean(vals)
    print(f"KVERITAS_METRIC name={name} value={g:.6g}")
    if mode == "full":
        print(f"KVERITAS_CLAIM metric={name} value={g:.6g}")
print(f"KVERITAS_METRIC name=rows_measured value={len(rows)}")
PY
