{
  "accuracy_by_question_type": {
    "knowledge-update": {
      "accuracy": 0.7179,
      "correct": 56,
      "total": 78
    },
    "multi-session": {
      "accuracy": 0.391,
      "correct": 52,
      "total": 133
    },
    "single-session-assistant": {
      "accuracy": 0.9286,
      "correct": 52,
      "total": 56
    },
    "single-session-preference": {
      "accuracy": 0.5667,
      "correct": 17,
      "total": 30
    },
    "single-session-user": {
      "accuracy": 0.9429,
      "correct": 66,
      "total": 70
    },
    "temporal-reasoning": {
      "accuracy": 0.4887,
      "correct": 65,
      "total": 133
    }
  },
  "benchmark": "LongMemEval",
  "dataset": {
    "path_name": "longmemeval_oracle.json",
    "questions_completed": 500,
    "questions_requested": 500,
    "sha256": "821a2034d219ab45846873dd14c14f12cfe7776e73527a483f9dac095d38620c"
  },
  "evaluation_status": "provisional_local_reader_and_local_judge",
  "limitations": [
    "The oracle split contains only evidence sessions and is easier than LongMemEval-S.",
    "The reader and judge are the same local model, so the QA score is provisional, not an official independent-judge result.",
    "Dataset timestamps simulate elapsed time; they are not years of wall-clock operation."
  ],
  "measured": {
    "diagnostic_exact_containment": 0.51,
    "false_recall_rate_on_abstention": 0.6,
    "median_total_latency_ms": 7075.98,
    "provisional_qa_accuracy": 0.616,
    "retrieval_session_recall_at_k": 1.0,
    "temporal_reasoning_accuracy": 0.4887
  },
  "models": {
    "judge": "qwen3.5:9b",
    "judge_independent_of_reader": false,
    "reader": "qwen3.5:9b"
  },
  "raw_log": "raw.jsonl",
  "run_id": "3bf5e3d4b579",
  "schema": "eliza.evidence.longmemeval.v1",
  "split": "official_cleaned_oracle",
  "started_at": "2026-09-20T09:05:59Z",
  "status": "complete",
  "updated_at": "2026-09-20T13:27:45Z"
}
