{
  "accuracy_by_question_type": {
    "knowledge-update": {
      "accuracy": 0.7821,
      "correct": 61,
      "total": 78
    },
    "multi-session": {
      "accuracy": 0.391,
      "correct": 52,
      "total": 133
    },
    "single-session-assistant": {
      "accuracy": 0.9286,
      "correct": 52,
      "total": 56
    },
    "single-session-preference": {
      "accuracy": 0.5,
      "correct": 15,
      "total": 30
    },
    "single-session-user": {
      "accuracy": 0.9286,
      "correct": 65,
      "total": 70
    },
    "temporal-reasoning": {
      "accuracy": 0.4586,
      "correct": 61,
      "total": 133
    }
  },
  "benchmark": "LongMemEval",
  "evaluation_status": "cross_model_local_judge_not_external_audit",
  "limitations": [
    "This regrades fixed published answers; it does not rerun retrieval or generation.",
    "The judge is a distinct local model, reducing same-model circularity but not constituting an external independent audit.",
    "The oracle split is easier than LongMemEval-S and its timestamps are synthetic rather than elapsed wall-clock years."
  ],
  "measured": {
    "agreement_with_original_judge": 0.944,
    "cross_model_qa_accuracy": 0.612,
    "false_recall_rate_on_abstention": 0.3667,
    "median_judge_latency_ms": 391.71
  },
  "models": {
    "cross_model_judge": "qwen3.8:latest",
    "externally_independent_audit": false,
    "judge_distinct_from_reader": true,
    "original_judge": "qwen3.5:9b",
    "reader": "qwen3.5:9b"
  },
  "raw_log": "raw.jsonl",
  "schema": "eliza.evidence.longmemeval.cross_model_judge.v1",
  "source": {
    "questions_completed": 500,
    "questions_requested": 500,
    "raw_log": "raw.jsonl",
    "sha256": "bfc31711c51c62c9df752c641e6f24decb1e66c3bac74b291196cc055885aeaa"
  },
  "started_at": "2026-09-20T15:25:05Z",
  "status": "complete",
  "updated_at": "2026-09-20T16:30:10Z"
}
