{
  "schema": "eliza.evidence.intelligence_index.methodology.v1",
  "name": "Eliza Intelligence Index",
  "short_name": "EII",
  "version": "0.1.0",
  "purpose": "A reproducible, coverage-gated view of measured Eliza system capabilities. It is not a human IQ test and does not claim consciousness, AGI or human equivalence.",
  "score_scale": {
    "minimum": 0,
    "maximum": 100,
    "domain_transform": "100 multiplied by the benchmark's primary accuracy or task-success rate",
    "missing_domains": "Never imputed as zero or success. They remain null and prevent publication of an overall score until the coverage gate is met.",
    "observed_domain_score": "Weighted mean of measured domains only. This diagnostic number must always be displayed beside its observed and standard coverage.",
    "overall_score": "Issued only when standard benchmark coverage is at least 70%, every critical domain has a result, raw case logs are public and the release declares its judge/audit status."
  },
  "coverage_gate": {
    "minimum_standard_coverage_percent": 70,
    "critical_domains": [
      "interactive_memory",
      "general_reasoning",
      "agentic_tool_use",
      "multimodal_understanding"
    ]
  },
  "domains": [
    {
      "id": "interactive_memory",
      "label": "Long-term interactive memory",
      "weight": 0.2,
      "primary_benchmark": "LongMemEval",
      "benchmark_url": "https://arxiv.org/abs/2410.10813",
      "critical": true
    },
    {
      "id": "conversation_context",
      "label": "Conversation and context integrity",
      "weight": 0.15,
      "primary_benchmark": "Versioned Eliza behavioural suite; diagnostic until an external benchmark is adopted",
      "critical": false
    },
    {
      "id": "general_reasoning",
      "label": "General knowledge and reasoning",
      "weight": 0.2,
      "primary_benchmark": "MMLU-Pro",
      "benchmark_url": "https://arxiv.org/abs/2406.01574",
      "critical": true
    },
    {
      "id": "agentic_tool_use",
      "label": "Agentic reasoning and tool use",
      "weight": 0.2,
      "primary_benchmark": "GAIA",
      "benchmark_url": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/25ae35b5b1738d80f1f03a8713e405ec-Abstract-Conference.html",
      "critical": true
    },
    {
      "id": "multimodal_understanding",
      "label": "Multimodal understanding",
      "weight": 0.15,
      "primary_benchmark": "MMMU",
      "benchmark_url": "https://openaccess.thecvf.com/content/CVPR2024/html/Yue_MMMU_A_Massive_Multi-discipline_Multimodal_Understanding_and_Reasoning_Benchmark_for_CVPR_2024_paper.html",
      "critical": true
    },
    {
      "id": "embodied_robotics",
      "label": "Embodied robotics",
      "weight": 0.1,
      "primary_benchmark": "Versioned physical task protocol with repeated trials, interventions and safety stops",
      "critical": false
    }
  ],
  "publication_rules": [
    "Publish the manifest, exact input hashes, raw per-case outputs, failures, timestamps, model declarations and limitations.",
    "Evaluate the deployed Eliza path where possible; component-only tests must be labelled as component diagnostics.",
    "Do not tune against held-out answers and then report the same cases as an unbiased test.",
    "Do not call synthetic elapsed time real elapsed time or local cross-model judging an independent external audit.",
    "Keep historical releases so regressions remain visible."
  ]
}
