{
  "title": "multivon-eval vs DeepEval vs RAGAS on hallucination detection",
  "last_run_date": "2026-06-26",
  "dataset": "ragtruth-sum",
  "cases": 100,
  "disclosure": "Recorded result. The standalone cross-framework harness is not public as of 2026-08-19, so this run is not labeled independently reproducible.",
  "judges": {
    "claude-haiku-4-5": [
      {"framework": "multivon-eval", "threshold": 0.9, "f1": 0.69, "precision": 0.615, "recall": 0.784, "errors": 0},
      {"framework": "DeepEval", "threshold": 0.5, "f1": 0.0, "precision": 0.0, "recall": 0.0, "errors": 100}
    ],
    "gpt-4o-mini": [
      {"framework": "multivon-eval", "threshold": 0.9, "f1": 0.729, "precision": 0.912, "recall": 0.608, "errors": 0},
      {"framework": "DeepEval", "threshold": 0.5, "f1": 0.038, "precision": 1.0, "recall": 0.02, "errors": 0},
      {"framework": "RAGAS", "threshold": 0.5, "f1": 0.038, "precision": 1.0, "recall": 0.02, "errors": 4}
    ]
  },
  "related_public_artifacts": "https://github.com/multivon-ai/multivon-eval/tree/main/benchmarks"
}
