{
  "title": "multivon-eval vs DeepEval vs RAGAS on hallucination detection",
  "last_run_date": "2026-06-26",
  "dataset": "ragtruth-sum",
  "cases": 100,
  "disclosure": "Recorded result. The standalone cross-framework harness is not public as of 2026-08-19, so this run is not labeled independently reproducible.",
  "judges": {
    "claude-haiku-4-5": [
      {
        "framework": "multivon-eval",
        "threshold": 0.9,
        "f1": 0.69,
        "precision": 0.615,
        "recall": 0.784,
        "errors": 0
      },
      {
        "framework": "DeepEval",
        "threshold": 0.5,
        "f1": null,
        "precision": null,
        "recall": null,
        "errors": 100,
        "status": "unmeasured: all cases errored"
      }
    ],
    "gpt-4o-mini": [
      {
        "framework": "multivon-eval",
        "threshold": 0.9,
        "f1": 0.729,
        "precision": 0.912,
        "recall": 0.608,
        "errors": 0
      },
      {
        "framework": "DeepEval",
        "threshold": 0.5,
        "f1": 0.038,
        "precision": 1.0,
        "recall": 0.02,
        "errors": 0
      },
      {
        "framework": "RAGAS",
        "threshold": 0.5,
        "f1": 0.038,
        "precision": 1.0,
        "recall": 0.02,
        "errors": 4
      }
    ]
  },
  "related_public_artifacts": "https://github.com/multivon-ai/multivon-eval/tree/main/benchmarks"
}
