#!/usr/bin/env bash
# examples/evaluation/beir: BAAI/bge-base-en-v1.5 dense retrieval (no reranker) on BEIR SciFact, NFCorpus and
# FiQA test sets, one GPU. Corpus embeddings are recomputed in this run. Scores parsed from the markdown it writes.
set -o pipefail
ROW=/var/tmp/kv-repro/038-FlagOpen-FlagEmbedding
export HOME=$ROW/data/home
T=$(date +%s)
cd "$ROW/data" || exit 1
echo "KVERITAS_PHASE name=evaluate"
python -m FlagEmbedding.evaluation.beir --eval_name beir --dataset_dir ./beir/data \
  --dataset_names scifact nfcorpus fiqa --splits test \
  --corpus_embd_save_dir ./beir/embd-$T --output_dir ./beir/results-$T --search_top_k 1000 \
  --k_values 10 100 --eval_output_method markdown --eval_output_path ./beir/results-$T.md \
  --eval_metrics ndcg_at_10 recall_at_100 --ignore_identical_ids True \
  --embedder_name_or_path BAAI/bge-base-en-v1.5 --devices cuda:0 2>&1 | grep -vE "it/s\]|s/it\]" | tail -5 || exit 1
cat ./beir/results-$T.md
python - ./beir/results-$T.md <<'PY'
import re, sys
text = open(sys.argv[1]).read()
for metric, body in re.findall(r"## (\w+)\n\n(.*?)(?=\n## |\Z)", text, re.S):
    lines = [l for l in body.splitlines() if l.startswith("|")]
    head = [c.strip() for c in lines[0].strip("|").split("|")]
    vals = [c.strip().strip("*") for c in lines[2].strip("|").split("|")]
    for name, v in zip(head[2:], vals[2:]):
        key = re.sub(r"\W+", "_", name.replace("-test", "")).strip("_") + "_" + metric
        print(f"KVERITAS_METRIC name={key} value={v}")
        print(f"KVERITAS_CLAIM metric={key} value={v}")
PY
