{
  "critical_domains_missing": [
    "general_reasoning",
    "agentic_tool_use",
    "multimodal_understanding"
  ],
  "domains": [
    {
      "benchmark_url": "https://arxiv.org/abs/2410.10813",
      "critical": true,
      "deployed_system_path": true,
      "evidence": "/evidence/longmemeval-cross-model-judge/latest.json",
      "externally_audited": false,
      "id": "interactive_memory",
      "label": "Long-term interactive memory",
      "limitations": [
        "This regrades fixed published answers; it does not rerun retrieval or generation.",
        "The judge is a distinct local model, reducing same-model circularity but not constituting an external independent audit.",
        "The oracle split is easier than LongMemEval-S and its timestamps are synthetic rather than elapsed wall-clock years."
      ],
      "primary_benchmark": "LongMemEval",
      "raw_log": "/evidence/longmemeval-cross-model-judge/raw.jsonl",
      "sample_count": 500,
      "score": 61.2,
      "sha256": "4aa5d8c0556b9071deadee93688e99d1dbd06143880cf51b82f144ec496c7139",
      "standard_benchmark": true,
      "status": "measured_provisional_local_cross_model_judge",
      "weight": 0.2
    },
    {
      "critical": false,
      "critical_failures": 3,
      "deployed_system_path": false,
      "evidence": "/evidence/intelligence-index/behavioral-model.json",
      "externally_audited": false,
      "id": "conversation_context",
      "label": "Conversation and context integrity",
      "limitations": [
        "This is a small fixed Eliza-specific suite, not an external standard benchmark.",
        "It exercises Brain's configured conversation model directly rather than every layer of the deployed Eliza response path.",
        "The score is diagnostic and is published with all failures; it cannot establish general intelligence."
      ],
      "passed_cases": 5,
      "primary_benchmark": "Versioned Eliza behavioural suite; diagnostic until an external benchmark is adopted",
      "sample_count": 10,
      "score": 50.0,
      "sha256": "d7fb4a089d595ff8f980b537faea1568f12e138fc1da37941ccc974aa3c10ba2",
      "standard_benchmark": false,
      "status": "measured_component_diagnostic",
      "weight": 0.15
    },
    {
      "benchmark_url": "https://arxiv.org/abs/2406.01574",
      "critical": true,
      "deployed_system_path": false,
      "externally_audited": false,
      "id": "general_reasoning",
      "label": "General knowledge and reasoning",
      "primary_benchmark": "MMLU-Pro",
      "sample_count": 0,
      "score": null,
      "standard_benchmark": false,
      "status": "not_measured",
      "weight": 0.2
    },
    {
      "benchmark_url": "https://proceedings.iclr.cc/paper_files/paper/2024/hash/25ae35b5b1738d80f1f03a8713e405ec-Abstract-Conference.html",
      "critical": true,
      "deployed_system_path": false,
      "externally_audited": false,
      "id": "agentic_tool_use",
      "label": "Agentic reasoning and tool use",
      "primary_benchmark": "GAIA",
      "sample_count": 0,
      "score": null,
      "standard_benchmark": false,
      "status": "not_measured",
      "weight": 0.2
    },
    {
      "benchmark_url": "https://openaccess.thecvf.com/content/CVPR2024/html/Yue_MMMU_A_Massive_Multi-discipline_Multimodal_Understanding_and_Reasoning_Benchmark_for_CVPR_2024_paper.html",
      "critical": true,
      "deployed_system_path": false,
      "externally_audited": false,
      "id": "multimodal_understanding",
      "label": "Multimodal understanding",
      "primary_benchmark": "MMMU",
      "sample_count": 0,
      "score": null,
      "standard_benchmark": false,
      "status": "not_measured",
      "weight": 0.15
    },
    {
      "critical": false,
      "deployed_system_path": false,
      "externally_audited": false,
      "id": "embodied_robotics",
      "label": "Embodied robotics",
      "primary_benchmark": "Versioned physical task protocol with repeated trials, interventions and safety stops",
      "sample_count": 0,
      "score": null,
      "standard_benchmark": false,
      "status": "not_measured",
      "weight": 0.1
    }
  ],
  "generated_at": "2026-10-04T21:59:19.967989Z",
  "interpretation": {
    "may_say": "On the currently observed domains, Eliza's evidence-weighted score is 56.40/100 with 35% observed coverage.",
    "must_also_say": "Only 20% has standard-benchmark coverage; no overall Eliza Intelligence Index score is issued yet.",
    "must_not_say": "This is Eliza's IQ, proof of AGI, human equivalence or a complete measure of intelligence."
  },
  "methodology": "/evidence/intelligence-index/methodology.json",
  "methodology_sha256": "e95d4a852010c5a2ac9fda8d5263ed43b13c69c1a2cff87865b8642ecaf0c832",
  "minimum_standard_coverage_percent": 70.0,
  "name": "Eliza Intelligence Index",
  "observed_coverage_percent": 35.0,
  "observed_domain_score": 56.4,
  "overall_score": null,
  "release_status": "preview_coverage_gated",
  "schema": "eliza.evidence.intelligence_index.result.v1",
  "standard_benchmark_coverage_percent": 20.0,
  "version": "0.1.0"
}
