{
  "schemaVersion": 2,
  "datasetVersion": "2026-08-16.2",
  "generatedAt": "2026-08-16T16:19:00.000Z",
  "fixture": false,
  "methodologyVersion": "resilient-frontier-index-v2",
  "sourceHierarchy": [
    "independent",
    "benchmark_org",
    "research_paper",
    "official_report",
    "third_party",
    "vendor"
  ],
  "categories": [
    {
      "id": "overview",
      "label": "Overview",
      "description": "Broad, composite evaluations intended for orientation rather than a universal score.",
      "order": 10
    },
    {
      "id": "professional-work",
      "label": "Professional work",
      "description": "Document, legal, financial, and real-world knowledge work.",
      "order": 20
    },
    {
      "id": "mathematics",
      "label": "Mathematics",
      "description": "Competition, olympiad, and frontier mathematical reasoning.",
      "order": 30
    },
    {
      "id": "science",
      "label": "Science",
      "description": "Scientific question answering and research reasoning.",
      "order": 40
    },
    {
      "id": "physics",
      "label": "Physics",
      "description": "Physics reasoning, modeling, and scientific discovery.",
      "order": 50
    },
    {
      "id": "software-engineering",
      "label": "Software engineering",
      "description": "Repository-level software engineering and issue resolution.",
      "order": 60
    },
    {
      "id": "coding-computation",
      "label": "Coding & computation",
      "description": "Code generation, execution, and competitive programming.",
      "order": 70
    },
    {
      "id": "agents-tool-use",
      "label": "Agents & tool use",
      "description": "Tool-mediated, browser, terminal, and multi-step task completion.",
      "order": 80
    },
    {
      "id": "novel-reasoning",
      "label": "Novel reasoning",
      "description": "Generalization and reasoning on unfamiliar problem structures.",
      "order": 90
    },
    {
      "id": "multimodal",
      "label": "Multimodal",
      "description": "Visual, diagram, document, and mixed-modality reasoning.",
      "order": 100
    },
    {
      "id": "factuality-calibration",
      "label": "Factuality & knowledge reliability",
      "description": "Factual reliability, abstention, and grounded knowledge.",
      "order": 110
    },
    {
      "id": "instruction-following",
      "label": "Instruction following",
      "description": "Adherence to complex, hierarchical, and multi-constraint instructions.",
      "order": 120
    },
    {
      "id": "deployment",
      "label": "Deployment",
      "description": "Operational measurements where lower, higher, or no ranking may be appropriate.",
      "order": 130
    }
  ],
  "models": [
    {
      "id": "claude-opus-5-max",
      "name": "Claude Opus 5",
      "shortName": "Opus 5 · Max",
      "provider": "Anthropic",
      "modelVersion": "claude-opus-5",
      "configuration": "Adaptive thinking · max effort",
      "releaseDate": "2026-07-24T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 10,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 25,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; effort changes token volume, not unit price."
      }
    },
    {
      "id": "claude-opus-5-xhigh",
      "name": "Claude Opus 5",
      "shortName": "Opus 5 · XHigh",
      "provider": "Anthropic",
      "modelVersion": "claude-opus-5",
      "configuration": "Adaptive thinking · xhigh effort",
      "releaseDate": "2026-07-24T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 20,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 25,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; effort changes token volume, not unit price."
      }
    },
    {
      "id": "claude-opus-5-high",
      "name": "Claude Opus 5",
      "shortName": "Opus 5 · High",
      "provider": "Anthropic",
      "modelVersion": "claude-opus-5",
      "configuration": "Adaptive thinking · high effort",
      "releaseDate": "2026-07-24T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 30,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 25,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; effort changes token volume, not unit price."
      }
    },
    {
      "id": "claude-opus-5-medium",
      "name": "Claude Opus 5",
      "shortName": "Opus 5 · Medium",
      "provider": "Anthropic",
      "modelVersion": "claude-opus-5",
      "configuration": "Adaptive thinking · medium effort",
      "releaseDate": "2026-07-24T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 40,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 25,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; effort changes token volume, not unit price."
      }
    },
    {
      "id": "claude-opus-5-low",
      "name": "Claude Opus 5",
      "shortName": "Opus 5 · Low",
      "provider": "Anthropic",
      "modelVersion": "claude-opus-5",
      "configuration": "Adaptive thinking · low effort",
      "releaseDate": "2026-07-24T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 50,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 25,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; effort changes token volume, not unit price."
      }
    },
    {
      "id": "claude-fable-5-low",
      "name": "Claude Fable 5",
      "shortName": "Fable 5 · Low",
      "provider": "Anthropic",
      "modelVersion": "claude-fable-5",
      "configuration": "Adaptive thinking · low effort · Opus 4.8 fallback",
      "releaseDate": "2026-06-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 104,
      "publicPricing": {
        "inputPerMillionTokens": 10,
        "outputPerMillionTokens": 50,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; source reports fallback behavior on some safety-classified requests."
      },
      "notes": "Adaptive thinking is always enabled; source reports Opus 4.8 fallback behavior on some safety-classified requests."
    },
    {
      "id": "claude-fable-5-medium",
      "name": "Claude Fable 5",
      "shortName": "Fable 5 · Medium",
      "provider": "Anthropic",
      "modelVersion": "claude-fable-5",
      "configuration": "Adaptive thinking · medium effort · Opus 4.8 fallback",
      "releaseDate": "2026-06-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 103,
      "publicPricing": {
        "inputPerMillionTokens": 10,
        "outputPerMillionTokens": 50,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; source reports fallback behavior on some safety-classified requests."
      },
      "notes": "Adaptive thinking is always enabled; source reports Opus 4.8 fallback behavior on some safety-classified requests."
    },
    {
      "id": "claude-fable-5-high",
      "name": "Claude Fable 5",
      "shortName": "Fable 5 · High",
      "provider": "Anthropic",
      "modelVersion": "claude-fable-5",
      "configuration": "Adaptive thinking · high effort · Opus 4.8 fallback",
      "releaseDate": "2026-06-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 102,
      "publicPricing": {
        "inputPerMillionTokens": 10,
        "outputPerMillionTokens": 50,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; source reports fallback behavior on some safety-classified requests."
      },
      "notes": "Adaptive thinking is always enabled; source reports Opus 4.8 fallback behavior on some safety-classified requests."
    },
    {
      "id": "claude-fable-5-xhigh",
      "name": "Claude Fable 5",
      "shortName": "Fable 5 · Extra High",
      "provider": "Anthropic",
      "modelVersion": "claude-fable-5",
      "configuration": "Adaptive thinking · extra high effort · Opus 4.8 fallback",
      "releaseDate": "2026-06-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 101,
      "publicPricing": {
        "inputPerMillionTokens": 10,
        "outputPerMillionTokens": 50,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; source reports fallback behavior on some safety-classified requests."
      },
      "notes": "Adaptive thinking is always enabled; source reports Opus 4.8 fallback behavior on some safety-classified requests."
    },
    {
      "id": "claude-fable-5-max",
      "name": "Claude Fable 5",
      "shortName": "Fable 5 · Max",
      "provider": "Anthropic",
      "modelVersion": "claude-fable-5",
      "configuration": "Adaptive thinking · max effort · Opus 4.8 fallback",
      "releaseDate": "2026-06-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 100,
      "publicPricing": {
        "inputPerMillionTokens": 10,
        "outputPerMillionTokens": 50,
        "currency": "USD",
        "sourceUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard uncached API pricing; source reports fallback behavior on some safety-classified requests."
      },
      "notes": "Adaptive thinking is always enabled; source reports Opus 4.8 fallback behavior on some safety-classified requests."
    },
    {
      "id": "gpt-5-6-sol-none",
      "name": "GPT-5.6 Sol",
      "shortName": "GPT-5.6 Sol · None",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-sol",
      "configuration": "Reasoning · none effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 205,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 30,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-sol-low",
      "name": "GPT-5.6 Sol",
      "shortName": "GPT-5.6 Sol · Low",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-sol",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 204,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 30,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-sol-medium",
      "name": "GPT-5.6 Sol",
      "shortName": "GPT-5.6 Sol · Medium",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-sol",
      "configuration": "Reasoning · medium effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 203,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 30,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-sol-high",
      "name": "GPT-5.6 Sol",
      "shortName": "GPT-5.6 Sol · High",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-sol",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 202,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 30,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-sol-xhigh",
      "name": "GPT-5.6 Sol",
      "shortName": "GPT-5.6 Sol · Extra High",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-sol",
      "configuration": "Reasoning · extra high effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 201,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 30,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-sol-max",
      "name": "GPT-5.6 Sol",
      "shortName": "GPT-5.6 Sol · Max",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-sol",
      "configuration": "Reasoning · max effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 200,
      "publicPricing": {
        "inputPerMillionTokens": 5,
        "outputPerMillionTokens": 30,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-terra-none",
      "name": "GPT-5.6 Terra",
      "shortName": "GPT-5.6 Terra · None",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-terra",
      "configuration": "Reasoning · none effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 305,
      "publicPricing": {
        "inputPerMillionTokens": 2.5,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-terra-low",
      "name": "GPT-5.6 Terra",
      "shortName": "GPT-5.6 Terra · Low",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-terra",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 304,
      "publicPricing": {
        "inputPerMillionTokens": 2.5,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-terra-medium",
      "name": "GPT-5.6 Terra",
      "shortName": "GPT-5.6 Terra · Medium",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-terra",
      "configuration": "Reasoning · medium effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 303,
      "publicPricing": {
        "inputPerMillionTokens": 2.5,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-terra-high",
      "name": "GPT-5.6 Terra",
      "shortName": "GPT-5.6 Terra · High",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-terra",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 302,
      "publicPricing": {
        "inputPerMillionTokens": 2.5,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-terra-xhigh",
      "name": "GPT-5.6 Terra",
      "shortName": "GPT-5.6 Terra · Extra High",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-terra",
      "configuration": "Reasoning · extra high effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 301,
      "publicPricing": {
        "inputPerMillionTokens": 2.5,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-terra-max",
      "name": "GPT-5.6 Terra",
      "shortName": "GPT-5.6 Terra · Max",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-terra",
      "configuration": "Reasoning · max effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 300,
      "publicPricing": {
        "inputPerMillionTokens": 2.5,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-terra",
        "updatedAt": "2026-08-15T00:00:00.000Z"
      }
    },
    {
      "id": "gpt-5-6-luna-none",
      "name": "GPT-5.6 Luna",
      "shortName": "GPT-5.6 Luna · None",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-luna",
      "configuration": "Reasoning · none effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 405,
      "publicPricing": {
        "inputPerMillionTokens": 0.2,
        "outputPerMillionTokens": 1.2,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Current price after the July 30, 2026 price change."
      }
    },
    {
      "id": "gpt-5-6-luna-low",
      "name": "GPT-5.6 Luna",
      "shortName": "GPT-5.6 Luna · Low",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-luna",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 404,
      "publicPricing": {
        "inputPerMillionTokens": 0.2,
        "outputPerMillionTokens": 1.2,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Current price after the July 30, 2026 price change."
      }
    },
    {
      "id": "gpt-5-6-luna-medium",
      "name": "GPT-5.6 Luna",
      "shortName": "GPT-5.6 Luna · Medium",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-luna",
      "configuration": "Reasoning · medium effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 403,
      "publicPricing": {
        "inputPerMillionTokens": 0.2,
        "outputPerMillionTokens": 1.2,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Current price after the July 30, 2026 price change."
      }
    },
    {
      "id": "gpt-5-6-luna-high",
      "name": "GPT-5.6 Luna",
      "shortName": "GPT-5.6 Luna · High",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-luna",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 402,
      "publicPricing": {
        "inputPerMillionTokens": 0.2,
        "outputPerMillionTokens": 1.2,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Current price after the July 30, 2026 price change."
      }
    },
    {
      "id": "gpt-5-6-luna-xhigh",
      "name": "GPT-5.6 Luna",
      "shortName": "GPT-5.6 Luna · Extra High",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-luna",
      "configuration": "Reasoning · extra high effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 401,
      "publicPricing": {
        "inputPerMillionTokens": 0.2,
        "outputPerMillionTokens": 1.2,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Current price after the July 30, 2026 price change."
      }
    },
    {
      "id": "gpt-5-6-luna-max",
      "name": "GPT-5.6 Luna",
      "shortName": "GPT-5.6 Luna · Max",
      "provider": "OpenAI",
      "modelVersion": "gpt-5.6-luna",
      "configuration": "Reasoning · max effort",
      "releaseDate": "2026-07-09T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://developers.openai.com/api/docs/guides/reasoning",
      "defaultOrder": 400,
      "publicPricing": {
        "inputPerMillionTokens": 0.2,
        "outputPerMillionTokens": 1.2,
        "currency": "USD",
        "sourceUrl": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Current price after the July 30, 2026 price change."
      }
    },
    {
      "id": "grok-4-6-low",
      "name": "Grok 4.6",
      "shortName": "Grok 4.6 · Low",
      "provider": "SpaceXAI (xAI)",
      "modelVersion": "grok-4.6",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-08-12T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://docs.x.ai/developers/model-capabilities/text/reasoning",
      "defaultOrder": 504,
      "publicPricing": {
        "inputPerMillionTokens": 2,
        "outputPerMillionTokens": 6,
        "currency": "USD",
        "sourceUrl": "https://docs.x.ai/docs/models",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Base tier below 200k prompt tokens; higher long-context tier is documented separately."
      }
    },
    {
      "id": "grok-4-6-medium",
      "name": "Grok 4.6",
      "shortName": "Grok 4.6 · Medium",
      "provider": "SpaceXAI (xAI)",
      "modelVersion": "grok-4.6",
      "configuration": "Reasoning · medium effort",
      "releaseDate": "2026-08-12T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://docs.x.ai/developers/model-capabilities/text/reasoning",
      "defaultOrder": 503,
      "publicPricing": {
        "inputPerMillionTokens": 2,
        "outputPerMillionTokens": 6,
        "currency": "USD",
        "sourceUrl": "https://docs.x.ai/docs/models",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Base tier below 200k prompt tokens; higher long-context tier is documented separately."
      }
    },
    {
      "id": "grok-4-6-high",
      "name": "Grok 4.6",
      "shortName": "Grok 4.6 · High",
      "provider": "SpaceXAI (xAI)",
      "modelVersion": "grok-4.6",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-08-12T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://docs.x.ai/developers/model-capabilities/text/reasoning",
      "defaultOrder": 502,
      "publicPricing": {
        "inputPerMillionTokens": 2,
        "outputPerMillionTokens": 6,
        "currency": "USD",
        "sourceUrl": "https://docs.x.ai/docs/models",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Base tier below 200k prompt tokens; higher long-context tier is documented separately."
      }
    },
    {
      "id": "grok-4-6-xhigh",
      "name": "Grok 4.6",
      "shortName": "Grok 4.6 · Extra High",
      "provider": "SpaceXAI (xAI)",
      "modelVersion": "grok-4.6",
      "configuration": "Reasoning · extra high effort",
      "releaseDate": "2026-08-12T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://docs.x.ai/developers/model-capabilities/text/reasoning",
      "defaultOrder": 501,
      "publicPricing": {
        "inputPerMillionTokens": 2,
        "outputPerMillionTokens": 6,
        "currency": "USD",
        "sourceUrl": "https://docs.x.ai/docs/models",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Base tier below 200k prompt tokens; higher long-context tier is documented separately."
      }
    },
    {
      "id": "kimi-k3-low",
      "name": "Kimi K3",
      "shortName": "Kimi K3 · Low",
      "provider": "Moonshot AI",
      "modelVersion": "kimi-k3",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-07-16T00:00:00.000Z",
      "current": true,
      "availability": "open_weights",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://platform.kimi.ai/docs/guide/use-reasoning-effort",
      "defaultOrder": 604,
      "publicPricing": {
        "inputPerMillionTokens": 3,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://platform.kimi.ai/docs/pricing/chat-k3",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Hosted API cache-miss pricing; self-hosted weights are a separate deployment."
      }
    },
    {
      "id": "kimi-k3-high",
      "name": "Kimi K3",
      "shortName": "Kimi K3 · High",
      "provider": "Moonshot AI",
      "modelVersion": "kimi-k3",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-07-16T00:00:00.000Z",
      "current": true,
      "availability": "open_weights",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://platform.kimi.ai/docs/guide/use-reasoning-effort",
      "defaultOrder": 602,
      "publicPricing": {
        "inputPerMillionTokens": 3,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://platform.kimi.ai/docs/pricing/chat-k3",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Hosted API cache-miss pricing; self-hosted weights are a separate deployment."
      }
    },
    {
      "id": "kimi-k3-max",
      "name": "Kimi K3",
      "shortName": "Kimi K3 · Max",
      "provider": "Moonshot AI",
      "modelVersion": "kimi-k3",
      "configuration": "Reasoning · max effort",
      "releaseDate": "2026-07-16T00:00:00.000Z",
      "current": true,
      "availability": "open_weights",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://platform.kimi.ai/docs/guide/use-reasoning-effort",
      "defaultOrder": 600,
      "publicPricing": {
        "inputPerMillionTokens": 3,
        "outputPerMillionTokens": 15,
        "currency": "USD",
        "sourceUrl": "https://platform.kimi.ai/docs/pricing/chat-k3",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Hosted API cache-miss pricing; self-hosted weights are a separate deployment."
      }
    },
    {
      "id": "gemini-3-1-pro-low",
      "name": "Gemini 3.1 Pro Preview",
      "shortName": "Gemini 3.1 Pro · Low",
      "provider": "Google",
      "modelVersion": "gemini-3.1-pro-preview",
      "configuration": "Thinking · low effort",
      "releaseDate": "2026-02-19T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "thinking_level",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://ai.google.dev/gemini-api/docs/generate-content/thinking",
      "defaultOrder": 704,
      "publicPricing": {
        "inputPerMillionTokens": 2,
        "outputPerMillionTokens": 12,
        "currency": "USD",
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard tier up to 200k input tokens; long-context pricing is documented separately."
      }
    },
    {
      "id": "gemini-3-1-pro-medium",
      "name": "Gemini 3.1 Pro Preview",
      "shortName": "Gemini 3.1 Pro · Medium",
      "provider": "Google",
      "modelVersion": "gemini-3.1-pro-preview",
      "configuration": "Thinking · medium effort",
      "releaseDate": "2026-02-19T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "thinking_level",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://ai.google.dev/gemini-api/docs/generate-content/thinking",
      "defaultOrder": 703,
      "publicPricing": {
        "inputPerMillionTokens": 2,
        "outputPerMillionTokens": 12,
        "currency": "USD",
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard tier up to 200k input tokens; long-context pricing is documented separately."
      }
    },
    {
      "id": "gemini-3-1-pro-high",
      "name": "Gemini 3.1 Pro Preview",
      "shortName": "Gemini 3.1 Pro · High",
      "provider": "Google",
      "modelVersion": "gemini-3.1-pro-preview",
      "configuration": "Thinking · high effort",
      "releaseDate": "2026-02-19T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "thinking_level",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://ai.google.dev/gemini-api/docs/generate-content/thinking",
      "defaultOrder": 702,
      "publicPricing": {
        "inputPerMillionTokens": 2,
        "outputPerMillionTokens": 12,
        "currency": "USD",
        "sourceUrl": "https://ai.google.dev/gemini-api/docs/pricing",
        "updatedAt": "2026-08-15T00:00:00.000Z",
        "notes": "Standard tier up to 200k input tokens; long-context pricing is documented separately."
      }
    },
    {
      "id": "deepseek-v4-pro-low",
      "name": "DeepSeek V4 Pro",
      "shortName": "DeepSeek V4 Pro · Low",
      "provider": "DeepSeek",
      "modelVersion": "deepseek-v4-pro",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://api-docs.deepseek.com/guides/thinking_mode/",
      "defaultOrder": 804
    },
    {
      "id": "deepseek-v4-pro-high",
      "name": "DeepSeek V4 Pro",
      "shortName": "DeepSeek V4 Pro · High",
      "provider": "DeepSeek",
      "modelVersion": "deepseek-v4-pro",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://api-docs.deepseek.com/guides/thinking_mode/",
      "defaultOrder": 802
    },
    {
      "id": "deepseek-v4-pro-max",
      "name": "DeepSeek V4 Pro",
      "shortName": "DeepSeek V4 Pro · Max",
      "provider": "DeepSeek",
      "modelVersion": "deepseek-v4-pro",
      "configuration": "Reasoning · max effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://api-docs.deepseek.com/guides/thinking_mode/",
      "defaultOrder": 800
    },
    {
      "id": "qwen-3-8-max-low",
      "name": "Qwen3.8-Max",
      "shortName": "Qwen3.8-Max · Low",
      "provider": "Alibaba / Qwen",
      "modelVersion": "qwen3.8-max",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "xhigh"
      ],
      "reasoningEffortDefault": "xhigh",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://docs.qwencloud.com/developer-guides/text-generation/thinking",
      "defaultOrder": 904
    },
    {
      "id": "qwen-3-8-max-medium",
      "name": "Qwen3.8-Max",
      "shortName": "Qwen3.8-Max · Medium",
      "provider": "Alibaba / Qwen",
      "modelVersion": "qwen3.8-max",
      "configuration": "Reasoning · medium effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "xhigh"
      ],
      "reasoningEffortDefault": "xhigh",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://docs.qwencloud.com/developer-guides/text-generation/thinking",
      "defaultOrder": 903
    },
    {
      "id": "qwen-3-8-max-xhigh",
      "name": "Qwen3.8-Max",
      "shortName": "Qwen3.8-Max · Extra High",
      "provider": "Alibaba / Qwen",
      "modelVersion": "qwen3.8-max",
      "configuration": "Reasoning · extra high effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "xhigh"
      ],
      "reasoningEffortDefault": "xhigh",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://docs.qwencloud.com/developer-guides/text-generation/thinking",
      "defaultOrder": 901
    },
    {
      "id": "muse-spark-1-2-minimal",
      "name": "Muse Spark 1.2",
      "shortName": "Muse Spark 1.2 · Minimal",
      "provider": "Meta",
      "modelVersion": "muse-spark-1.2",
      "configuration": "Reasoning · minimal effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://dev.meta.ai/docs/features/reasoning",
      "defaultOrder": 1006
    },
    {
      "id": "muse-spark-1-2-low",
      "name": "Muse Spark 1.2",
      "shortName": "Muse Spark 1.2 · Low",
      "provider": "Meta",
      "modelVersion": "muse-spark-1.2",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://dev.meta.ai/docs/features/reasoning",
      "defaultOrder": 1004
    },
    {
      "id": "muse-spark-1-2-medium",
      "name": "Muse Spark 1.2",
      "shortName": "Muse Spark 1.2 · Medium",
      "provider": "Meta",
      "modelVersion": "muse-spark-1.2",
      "configuration": "Reasoning · medium effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://dev.meta.ai/docs/features/reasoning",
      "defaultOrder": 1003
    },
    {
      "id": "muse-spark-1-2-high",
      "name": "Muse Spark 1.2",
      "shortName": "Muse Spark 1.2 · High",
      "provider": "Meta",
      "modelVersion": "muse-spark-1.2",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://dev.meta.ai/docs/features/reasoning",
      "defaultOrder": 1002
    },
    {
      "id": "muse-spark-1-2-xhigh",
      "name": "Muse Spark 1.2",
      "shortName": "Muse Spark 1.2 · Extra High",
      "provider": "Meta",
      "modelVersion": "muse-spark-1.2",
      "configuration": "Reasoning · extra high effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://dev.meta.ai/docs/features/reasoning",
      "defaultOrder": 1001
    },
    {
      "id": "glm-5-2-none",
      "name": "GLM-5.2",
      "shortName": "GLM-5.2 · None",
      "provider": "Z.ai",
      "modelVersion": "glm-5.2",
      "configuration": "Reasoning · none effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://docs.z.ai/guides/capabilities/thinking",
      "defaultOrder": 1105
    },
    {
      "id": "glm-5-2-minimal",
      "name": "GLM-5.2",
      "shortName": "GLM-5.2 · Minimal",
      "provider": "Z.ai",
      "modelVersion": "glm-5.2",
      "configuration": "Reasoning · minimal effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://docs.z.ai/guides/capabilities/thinking",
      "defaultOrder": 1106
    },
    {
      "id": "glm-5-2-low",
      "name": "GLM-5.2",
      "shortName": "GLM-5.2 · Low",
      "provider": "Z.ai",
      "modelVersion": "glm-5.2",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://docs.z.ai/guides/capabilities/thinking",
      "defaultOrder": 1104
    },
    {
      "id": "glm-5-2-medium",
      "name": "GLM-5.2",
      "shortName": "GLM-5.2 · Medium",
      "provider": "Z.ai",
      "modelVersion": "glm-5.2",
      "configuration": "Reasoning · medium effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://docs.z.ai/guides/capabilities/thinking",
      "defaultOrder": 1103
    },
    {
      "id": "glm-5-2-high",
      "name": "GLM-5.2",
      "shortName": "GLM-5.2 · High",
      "provider": "Z.ai",
      "modelVersion": "glm-5.2",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://docs.z.ai/guides/capabilities/thinking",
      "defaultOrder": 1102
    },
    {
      "id": "glm-5-2-xhigh",
      "name": "GLM-5.2",
      "shortName": "GLM-5.2 · Extra High",
      "provider": "Z.ai",
      "modelVersion": "glm-5.2",
      "configuration": "Reasoning · extra high effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://docs.z.ai/guides/capabilities/thinking",
      "defaultOrder": 1101
    },
    {
      "id": "glm-5-2-max",
      "name": "GLM-5.2",
      "shortName": "GLM-5.2 · Max",
      "provider": "Z.ai",
      "modelVersion": "glm-5.2",
      "configuration": "Reasoning · max effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://docs.z.ai/guides/capabilities/thinking",
      "defaultOrder": 1100
    },
    {
      "id": "glm-5-3-max",
      "name": "GLM-5.3",
      "shortName": "GLM-5.3 · Max",
      "provider": "Z.ai",
      "modelVersion": "glm-5.3",
      "configuration": "Reasoning · max effort",
      "releaseDate": "2026-08-14T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "none",
        "minimal",
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "max",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://docs.z.ai/guides/llm/glm-5.3",
      "defaultOrder": 1099
    },
    {
      "id": "gemini-3-7-flash-low",
      "name": "Gemini 3.7 Flash",
      "shortName": "Gemini 3.7 Flash · Low",
      "provider": "Google",
      "modelVersion": "gemini-3.7-flash",
      "configuration": "Thinking · low effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "thinking_level",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://ai.google.dev/gemini-api/docs/generate-content/thinking",
      "defaultOrder": 1204
    },
    {
      "id": "gemini-3-7-flash-medium",
      "name": "Gemini 3.7 Flash",
      "shortName": "Gemini 3.7 Flash · Medium",
      "provider": "Google",
      "modelVersion": "gemini-3.7-flash",
      "configuration": "Thinking · medium effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "thinking_level",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://ai.google.dev/gemini-api/docs/generate-content/thinking",
      "defaultOrder": 1203
    },
    {
      "id": "gemini-3-7-flash-high",
      "name": "Gemini 3.7 Flash",
      "shortName": "Gemini 3.7 Flash · High",
      "provider": "Google",
      "modelVersion": "gemini-3.7-flash",
      "configuration": "Thinking · high effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high"
      ],
      "reasoningEffortDefault": "medium",
      "reasoningControl": "thinking_level",
      "thinkingCanBeDisabled": false,
      "reasoningSourceUrl": "https://ai.google.dev/gemini-api/docs/generate-content/thinking",
      "defaultOrder": 1202
    },
    {
      "id": "claude-sonnet-5-low",
      "name": "Claude Sonnet 5",
      "shortName": "Sonnet 5 · Low",
      "provider": "Anthropic",
      "modelVersion": "claude-sonnet-5",
      "configuration": "Adaptive thinking · low effort",
      "releaseDate": "2026-06-30T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 1304
    },
    {
      "id": "claude-sonnet-5-medium",
      "name": "Claude Sonnet 5",
      "shortName": "Sonnet 5 · Medium",
      "provider": "Anthropic",
      "modelVersion": "claude-sonnet-5",
      "configuration": "Adaptive thinking · medium effort",
      "releaseDate": "2026-06-30T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 1303
    },
    {
      "id": "claude-sonnet-5-high",
      "name": "Claude Sonnet 5",
      "shortName": "Sonnet 5 · High",
      "provider": "Anthropic",
      "modelVersion": "claude-sonnet-5",
      "configuration": "Adaptive thinking · high effort",
      "releaseDate": "2026-06-30T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 1302
    },
    {
      "id": "claude-sonnet-5-xhigh",
      "name": "Claude Sonnet 5",
      "shortName": "Sonnet 5 · Extra High",
      "provider": "Anthropic",
      "modelVersion": "claude-sonnet-5",
      "configuration": "Adaptive thinking · extra high effort",
      "releaseDate": "2026-06-30T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 1301
    },
    {
      "id": "claude-sonnet-5-max",
      "name": "Claude Sonnet 5",
      "shortName": "Sonnet 5 · Max",
      "provider": "Anthropic",
      "modelVersion": "claude-sonnet-5",
      "configuration": "Adaptive thinking · max effort",
      "releaseDate": "2026-06-30T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "medium",
        "high",
        "xhigh",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "output_config.effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://platform.claude.com/docs/en/build-with-claude/effort",
      "defaultOrder": 1300
    },
    {
      "id": "deepseek-v4-flash-low",
      "name": "DeepSeek V4 Flash",
      "shortName": "DeepSeek V4 Flash · Low",
      "provider": "DeepSeek",
      "modelVersion": "deepseek-v4-flash",
      "configuration": "Reasoning · low effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://api-docs.deepseek.com/guides/thinking_mode/",
      "defaultOrder": 1404
    },
    {
      "id": "deepseek-v4-flash-high",
      "name": "DeepSeek V4 Flash",
      "shortName": "DeepSeek V4 Flash · High",
      "provider": "DeepSeek",
      "modelVersion": "deepseek-v4-flash",
      "configuration": "Reasoning · high effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://api-docs.deepseek.com/guides/thinking_mode/",
      "defaultOrder": 1402
    },
    {
      "id": "deepseek-v4-flash-max",
      "name": "DeepSeek V4 Flash",
      "shortName": "DeepSeek V4 Flash · Max",
      "provider": "DeepSeek",
      "modelVersion": "deepseek-v4-flash",
      "configuration": "Reasoning · max effort",
      "releaseDate": "2026-08-13T00:00:00.000Z",
      "current": true,
      "availability": "api",
      "capabilities": [
        "reasoning",
        "coding",
        "agents",
        "multimodal"
      ],
      "reasoningEffortLevels": [
        "low",
        "high",
        "max"
      ],
      "reasoningEffortDefault": "high",
      "reasoningControl": "reasoning_effort",
      "thinkingCanBeDisabled": true,
      "reasoningSourceUrl": "https://api-docs.deepseek.com/guides/thinking_mode/",
      "defaultOrder": 1400
    }
  ],
  "audits": [
    {
      "id": "audit-gdpval-aa-v2",
      "benchmarkId": "gdpval-aa-v2",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "GDPval-AA v2",
      "description": "Agentic professional-work evaluation using blind pairwise comparisons converted to Elo ratings.",
      "domain": [
        "professional work",
        "knowledge work",
        "agents"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Elo rating",
      "scoreType": "elo",
      "unit": "elo",
      "precision": 0,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "included",
      "citationId": "citation-aa-gdpval",
      "methodology": "220 tasks across 44 occupations and 9 industries are evaluated in an agentic loop with shell access and web browsing. Blind pairwise judgments are converted into Elo ratings anchored to human experts at 1000.",
      "order": 10
    },
    {
      "id": "audit-gpqa-diamond",
      "benchmarkId": "gpqa-diamond",
      "auditState": "verified",
      "categoryId": "science",
      "label": "GPQA Diamond",
      "description": "Expert-level graduate science questions with a native accuracy metric.",
      "domain": [
        "biology",
        "physics",
        "chemistry"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "included",
      "citationId": "citation-aa-gpqa",
      "methodology": "198 expert-written multiple-choice questions are scored pass@1 with no tools. The benchmark page identifies the evaluation version and current leaderboard snapshot.",
      "order": 20,
      "coverageNote": "Target cohort coverage is generated from exact, caveated, and missing classifications; fallback and lower-effort rows remain visible and never silently become exact."
    },
    {
      "id": "audit-aa-intelligence-index",
      "benchmarkId": "aa-intelligence-index",
      "auditState": "verified",
      "categoryId": "overview",
      "label": "Artificial Analysis Intelligence Index",
      "description": "Composite index combining several Artificial Analysis benchmark families.",
      "domain": [
        "general intelligence"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/methodology/intelligence-benchmarking",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Artificial Analysis Intelligence Index",
      "scoreType": "custom",
      "unit": "score",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-aa-methodology",
      "methodology": "Composite index defined by Artificial Analysis from its versioned benchmark suite. It is tracked as a candidate because its transformed inputs must remain separate from native benchmark rows.",
      "order": 15,
      "coverageNote": "External composite/check only. It is displayed for context and excluded from TLS because its components overlap native benchmark rows.",
      "admissionStatus": "excluded",
      "admissionReason": "External composite excluded because its components overlap native benchmark rows."
    },
    {
      "id": "audit-prbench-legal",
      "benchmarkId": "prbench-legal",
      "auditState": "candidate",
      "categoryId": "professional-work",
      "label": "PRBench Legal",
      "description": "Legal subset of Scale's Professional Reasoning Benchmark, scored with weighted expert rubrics.",
      "domain": [
        "legal reasoning",
        "professional work"
      ],
      "organization": "Scale Labs",
      "benchmarkUrl": "https://labs.scale.com/leaderboard/prbench-legal",
      "benchmarkVersion": "Public leaderboard snapshot · unpinned",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Weighted rubric score",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-prbench-legal",
      "methodology": "500 legal tasks are assessed with detailed weighted rubrics co-designed with credentialed professionals.",
      "order": 30,
      "coverageNote": "Candidate only: the public board does not expose comparable results for all 11 requested configurations."
    },
    {
      "id": "audit-gdp-pdf",
      "benchmarkId": "gdp-pdf",
      "auditState": "candidate",
      "categoryId": "multimodal",
      "label": "GDP.pdf",
      "description": "Professional multimodal reasoning over real-world PDF tasks.",
      "domain": [
        "documents",
        "multimodal",
        "professional work"
      ],
      "organization": "Surge AI",
      "benchmarkUrl": "https://surgehq.ai/leaderboards/gdp-pdf",
      "benchmarkVersion": "Public leaderboard snapshot · unpinned",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-gdp-pdf",
      "methodology": "Professional PDF tasks are graded against task-specific success criteria. The public leaderboard does not expose a stable version string or complete configuration ladder.",
      "order": 40,
      "coverageNote": "Candidate only: six of eleven requested configurations are readable and the public page has no stable version pin."
    },
    {
      "id": "audit-facts-grounding-v2",
      "benchmarkId": "facts-grounding-v2",
      "auditState": "candidate",
      "categoryId": "factuality-calibration",
      "label": "FACTS Grounding v2",
      "description": "Long-form factual grounding against supplied documents.",
      "domain": [
        "grounding",
        "factuality"
      ],
      "organization": "Google Research",
      "benchmarkUrl": "https://www.kaggle.com/benchmarks/google/facts-grounding",
      "benchmarkVersion": "Grounding v2 · public leaderboard snapshot",
      "benchmarkVersionDate": "2025-12-12T00:00:00.000Z",
      "metricName": "Grounding score",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-facts-grounding",
      "methodology": "The FACTS suite evaluates factuality across multiple tracks; Grounding v2 evaluates whether long-form answers remain supported by supplied evidence.",
      "order": 50,
      "coverageNote": "Candidate only: the current public leaderboard is client-rendered and does not expose requested-model values in the ingestion environment."
    },
    {
      "id": "audit-simpleqa",
      "benchmarkId": "simpleqa",
      "auditState": "candidate",
      "categoryId": "factuality-calibration",
      "label": "SimpleQA",
      "description": "Short-form factuality benchmark with an accuracy metric.",
      "domain": [
        "factuality",
        "knowledge"
      ],
      "organization": "OpenAI",
      "benchmarkUrl": "https://www.kaggle.com/benchmarks/openai/simpleqa/leaderboard",
      "benchmarkVersion": "SimpleQA public leaderboard snapshot · unpinned",
      "benchmarkVersionDate": "2026-07-22T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-simpleqa",
      "methodology": "Short-form questions are scored against a reference answer under the benchmark's published exact-answer protocol.",
      "order": 60,
      "coverageNote": "Candidate only: the hosted leaderboard is client-rendered and current requested-model values are not extractable."
    },
    {
      "id": "audit-aa-omniscience-accuracy",
      "benchmarkId": "aa-omniscience-accuracy",
      "auditState": "candidate",
      "categoryId": "factuality-calibration",
      "label": "AA-Omniscience Accuracy",
      "description": "Correct-answer rate from the AA-Omniscience factual knowledge evaluation.",
      "domain": [
        "factuality",
        "knowledge",
        "abstention"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/omniscience",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-aa-omniscience",
      "methodology": "6,000 questions across 42 topics and 6 domains are graded once by a fixed judge. Accuracy is the proportion answered correctly.",
      "order": 70,
      "coverageNote": "Candidate until a structured API snapshot is captured and pinned for all 11 requested configurations."
    },
    {
      "id": "audit-aa-omniscience-index",
      "benchmarkId": "aa-omniscience-index",
      "auditState": "candidate",
      "categoryId": "factuality-calibration",
      "label": "AA-Omniscience Index",
      "description": "Signed knowledge-reliability index rewarding correct answers and penalizing hallucinated guesses.",
      "domain": [
        "factuality",
        "knowledge",
        "abstention"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/omniscience",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Omniscience Index",
      "scoreType": "custom",
      "unit": "score",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-aa-omniscience",
      "methodology": "The signed index rewards correct answers, penalizes hallucinated guesses, and treats appropriate abstention as neutral.",
      "order": 80,
      "coverageNote": "Candidate until a structured API snapshot is captured and pinned for all 11 requested configurations."
    },
    {
      "id": "audit-aa-omniscience-hallucination-rate",
      "benchmarkId": "aa-omniscience-hallucination-rate",
      "auditState": "candidate",
      "categoryId": "factuality-calibration",
      "label": "AA-Omniscience Hallucination Rate",
      "description": "Share of non-correct responses that are confidently wrong under AA-Omniscience.",
      "domain": [
        "factuality",
        "hallucination"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/omniscience",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Hallucination rate",
      "scoreType": "error",
      "unit": "percent",
      "precision": 1,
      "direction": "lower",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-aa-omniscience",
      "methodology": "Incorrect divided by incorrect plus partial plus not-attempted responses. This is displayed beside accuracy only and is not scored independently.",
      "order": 90,
      "coverageNote": "Display-only candidate: lower native values reward abstention-heavy weak systems and are not a standalone capability ranking."
    },
    {
      "id": "audit-longfact-safe",
      "benchmarkId": "longfact-safe",
      "auditState": "candidate",
      "categoryId": "factuality-calibration",
      "label": "LongFact + SAFE",
      "description": "Long-form factuality evaluated with supported-fact precision and recall-style metrics.",
      "domain": [
        "factuality",
        "long-form generation"
      ],
      "organization": "Google Research",
      "benchmarkUrl": "https://arxiv.org/abs/2403.18802",
      "benchmarkVersion": "LongFact and SAFE",
      "benchmarkVersionDate": "2024-03-28T00:00:00.000Z",
      "metricName": "Supported-fact precision",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-longfact",
      "methodology": "Long-form answers are decomposed into atomic claims and checked for support by the SAFE pipeline.",
      "order": 100,
      "coverageNote": "Candidate: benchmark and native metrics are real, but current requested-model coverage is not sufficient."
    },
    {
      "id": "audit-confidencebench",
      "benchmarkId": "confidencebench",
      "auditState": "candidate",
      "categoryId": "factuality-calibration",
      "label": "ConfidenceBench",
      "description": "Confidence calibration benchmark scored with Brier score.",
      "domain": [
        "calibration",
        "confidence"
      ],
      "organization": "ConfidenceBench",
      "benchmarkUrl": "https://confidencebench.com/",
      "benchmarkVersion": "Public leaderboard snapshot · unpinned",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Brier score",
      "scoreType": "custom",
      "unit": "score",
      "precision": 3,
      "direction": "lower",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-confidencebench",
      "methodology": "Models answer private multiple-choice questions and self-report confidence on a 1–10 scale; Brier score evaluates calibration.",
      "order": 110,
      "coverageNote": "Candidate: only one requested configuration appears on the public board."
    },
    {
      "id": "audit-tau3-banking",
      "benchmarkId": "tau3-banking",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "τ³-Banking",
      "description": "Versioned banking agent evaluation scored by backend state correctness.",
      "domain": [
        "agents",
        "tool use",
        "banking"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/tau3-banking",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-tau3-banking",
      "methodology": "The benchmark evaluates 97 banking tasks and scores the resulting backend database state rather than conversational quality.",
      "order": 120,
      "coverageNote": "Candidate: the versioned live snapshot does not expose a complete five-effort Opus ladder."
    },
    {
      "id": "audit-aa-ttft-p50-10k",
      "benchmarkId": "aa-ttft-p50-10k",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "Time to First Token (P50, 10k input)",
      "description": "Rolling P50 time to first token from the Artificial Analysis API performance harness.",
      "domain": [
        "latency",
        "deployment"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/methodology/performance-benchmarking",
      "benchmarkVersion": "API Performance Benchmarking v2.2.0",
      "benchmarkVersionDate": "2026-03-02T00:00:00.000Z",
      "metricName": "P50 time to first token",
      "scoreType": "latency",
      "unit": "seconds",
      "precision": 2,
      "direction": "lower",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-aa-performance",
      "methodology": "A 10,000-token input workload is run eight times daily; the published value is the trailing-72-hour median.",
      "order": 130,
      "coverageNote": "Candidate: rolling operational metric, never part of capability scoring."
    },
    {
      "id": "audit-aa-e2e-response-p50-10k",
      "benchmarkId": "aa-e2e-response-p50-10k",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "End-to-End Response Time (P50, 10k input)",
      "description": "Rolling P50 end-to-end response time from the Artificial Analysis API performance harness.",
      "domain": [
        "latency",
        "deployment"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/methodology/performance-benchmarking",
      "benchmarkVersion": "API Performance Benchmarking v2.2.0",
      "benchmarkVersionDate": "2026-03-02T00:00:00.000Z",
      "metricName": "P50 end-to-end response time",
      "scoreType": "latency",
      "unit": "seconds",
      "precision": 2,
      "direction": "lower",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-aa-performance",
      "methodology": "A 10,000-token input workload is run eight times daily; the published value is the trailing-72-hour median.",
      "order": 140,
      "coverageNote": "Candidate: rolling operational metric, never part of capability scoring."
    },
    {
      "id": "audit-frontiermath",
      "benchmarkId": "frontiermath",
      "auditState": "candidate",
      "categoryId": "mathematics",
      "label": "FrontierMath",
      "description": "Frontier mathematics evaluation with public benchmark documentation.",
      "domain": [
        "mathematics"
      ],
      "organization": "Epoch AI",
      "benchmarkUrl": "https://epoch.ai/frontiermath",
      "benchmarkVersion": "FrontierMath v2",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-frontiermath",
      "methodology": "Private and public mathematics tiers are scored by exact solution correctness.",
      "order": 150,
      "coverageNote": "Candidate: requested-model values are not sufficiently extractable and version-pinned."
    },
    {
      "id": "audit-aime-2025",
      "benchmarkId": "aime-2025",
      "auditState": "candidate",
      "categoryId": "mathematics",
      "label": "AIME 2025",
      "description": "American Invitational Mathematics Examination 2025 problem set.",
      "domain": [
        "mathematics"
      ],
      "organization": "American Invitational Mathematics Examination",
      "benchmarkUrl": "https://artofproblemsolving.com/wiki/index.php/AIME",
      "benchmarkVersion": "AIME 2025",
      "benchmarkVersionDate": "2025-03-01T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-aime",
      "methodology": "The examination has 15 integer-answer problems; accuracy is the share solved correctly.",
      "order": 160,
      "coverageNote": "Candidate: no current requested-model public result source was verified."
    },
    {
      "id": "audit-swe-bench-verified",
      "benchmarkId": "swe-bench-verified",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "SWE-bench Verified — Vals",
      "description": "Repository-level issue resolution on the Vals mini-SWE-agent SWE-bench Verified run.",
      "domain": [
        "software engineering",
        "coding",
        "repository repair"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/swebench",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Resolved issue rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-swebench",
      "methodology": "The requested values are reported under one named comparison family. Vals AI mini-SWE-agent bash harness Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 845,
      "coverageNote": "Requested max-configuration values are retained under Vals AI mini-SWE-agent bash harness; absent models remain blank.",
      "requiredScaffold": "Vals AI mini-SWE-agent bash harness",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-swe-bench-pro",
      "benchmarkId": "swe-bench-pro",
      "auditState": "candidate",
      "categoryId": "software-engineering",
      "label": "SWE-bench Pro",
      "description": "Professional software engineering benchmark with vendor-scaffold and neutral-harness variants.",
      "domain": [
        "software engineering",
        "coding"
      ],
      "organization": "SWE-bench",
      "benchmarkUrl": "https://www.swebench.com/",
      "benchmarkVersion": "Pro · public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Resolved issue rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-swebench",
      "methodology": "Repository-level issue patches are evaluated under a benchmark-defined professional task set.",
      "order": 180,
      "coverageNote": "Candidate: vendor scaffold differences make current cross-model figures limited-comparability."
    },
    {
      "id": "audit-livecodebench",
      "benchmarkId": "livecodebench",
      "auditState": "verified",
      "categoryId": "coding-computation",
      "label": "LiveCodeBench · Vals run",
      "description": "Contamination-resistant competitive coding evaluation using recent programming problems.",
      "domain": [
        "coding",
        "competitive programming",
        "reasoning"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/lcb",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass@1",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-livecodebench",
      "methodology": "The Vals LiveCodeBench run evaluates recent competition problems with a shared pass@1 harness. The Vals source is kept separate from official LiveCodeBench exports and from DeepSeek V4 rows that do not identify the requested Pro checkpoint.",
      "order": 790,
      "coverageNote": "The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing.",
      "requiredScaffold": "Vals AI LiveCodeBench harness",
      "admissionStatus": "excluded",
      "admissionReason": "Imported for transparent comparison only. The captured source does not establish the required frontier effort and/or has more than two missing target cells."
    },
    {
      "id": "audit-frontiercode",
      "benchmarkId": "frontiercode",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "FrontierCode v1.1 Extended",
      "description": "Cognition coding evaluation on the 150-task Extended private subset using the weighted rubric score.",
      "domain": [
        "coding",
        "software engineering"
      ],
      "organization": "Cognition",
      "benchmarkUrl": "https://cognition.com/blog/frontier-code",
      "benchmarkVersion": "FrontierCode v1.1 Extended",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Extended weighted score",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-frontiercode-11-data",
      "methodology": "Cognition runs five trials per available effort and reports the best effort average. The weighted score is distinct from the blocker-clearing pass rate.",
      "order": 895,
      "coverageNote": "Only exact Extended-subset weighted scores are retained; Main results and pass-rate figures are not substituted.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-terminal-bench-2",
      "benchmarkId": "terminal-bench-2",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "Terminal-Bench 2.1",
      "description": "Terminal task-completion benchmark under a versioned agent harness.",
      "domain": [
        "agents",
        "terminal use",
        "coding"
      ],
      "organization": "Terminal-Bench",
      "benchmarkUrl": "https://www.tbench.ai/",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-terminal-bench",
      "methodology": "The requested values are retained as an informative cross-source comparison. Version-matched public Terminal-Bench 2.1 results Public provenance does not establish one fully controlled harness, so the row is marked with a dagger and excluded from aggregate scoring.",
      "order": 910,
      "coverageNote": "Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-tau-bench",
      "benchmarkId": "tau-bench",
      "auditState": "rejected",
      "categoryId": "agents-tool-use",
      "label": "τ-bench (legacy generic label)",
      "description": "Legacy generic row without a pinned τ-bench family or version.",
      "domain": [
        "agents",
        "tool use"
      ],
      "organization": "Sierra Research",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/tau3-banking",
      "benchmarkVersion": "Unpinned legacy label",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "archived",
      "citationId": "citation-tau3-banking",
      "methodology": "Rejected because the legacy row does not identify τ², τ³, domain, grader, or snapshot.",
      "order": 220,
      "rejectionReason": "Generic label cannot establish a real versioned metric."
    },
    {
      "id": "audit-arc-agi-2",
      "benchmarkId": "arc-agi-2",
      "auditState": "candidate",
      "categoryId": "novel-reasoning",
      "label": "ARC-AGI-2",
      "description": "Novel abstract reasoning benchmark maintained by ARC Prize.",
      "domain": [
        "abstract reasoning",
        "generalization"
      ],
      "organization": "ARC Prize Foundation",
      "benchmarkUrl": "https://arcprize.org/",
      "benchmarkVersion": "ARC-AGI-2 · public results snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-arc-prize",
      "methodology": "The benchmark scores correct solutions on a semi-private task set maintained by ARC Prize.",
      "order": 230,
      "coverageNote": "Candidate: current requested-model configuration coverage is not complete and public values vary by effort."
    },
    {
      "id": "audit-mmmu-pro",
      "benchmarkId": "mmmu-pro",
      "auditState": "verified",
      "categoryId": "multimodal",
      "label": "MMMU-Pro · Vals run",
      "description": "Multimodal academic and professional reasoning evaluation under the MMMU-Pro protocol.",
      "domain": [
        "multimodal",
        "vision",
        "reasoning"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/mmmu",
      "benchmarkVersion": "Vals MMMU-Pro · public leaderboard snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-mmmu-pro",
      "methodology": "MMMU-Pro evaluates multimodal questions with fixed image handling and pass@1 scoring. This Vals record is kept separate from generic MMMU and other provider reports; missing model rows are not filled from a different image protocol.",
      "order": 780,
      "coverageNote": "The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified.",
      "requiredScaffold": "Vals AI MMMU-Pro evaluation harness",
      "admissionStatus": "excluded",
      "admissionReason": "Imported for transparent comparison only. The captured source does not establish the required frontier effort and/or has more than two missing target cells."
    },
    {
      "id": "audit-physunibench",
      "benchmarkId": "physunibench",
      "auditState": "candidate",
      "categoryId": "physics",
      "label": "PhysUniBench",
      "description": "Multimodal undergraduate physics reasoning benchmark with diagram-grounded questions.",
      "domain": [
        "physics",
        "multimodal",
        "scientific reasoning"
      ],
      "organization": "PrismaX Team",
      "benchmarkUrl": "https://prismax-team.github.io/PhysUniBenchmark/",
      "benchmarkVersion": "PhysUniBench v1",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-physunibench",
      "methodology": "3,304 human-verified multimodal physics problems cover eight undergraduate sub-disciplines in multiple-choice and open-ended formats. The source reports accuracy separately for the two response formats and sub-disciplines; no composite is inferred until a pinned aggregation is selected.",
      "order": 245,
      "coverageNote": "Real benchmark identity is now verified, but the public result tables do not yet provide a stable, effort-specific 67-configuration snapshot."
    },
    {
      "id": "audit-ifbench",
      "benchmarkId": "ifbench",
      "auditState": "candidate",
      "categoryId": "instruction-following",
      "label": "IFBench",
      "description": "Instruction-following benchmark with verifiable constraint checks.",
      "domain": [
        "instruction following"
      ],
      "organization": "IFBench",
      "benchmarkUrl": "https://huggingface.co/papers/2410.16248",
      "benchmarkVersion": "IFBench · public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Instruction-following accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-aa-methodology",
      "methodology": "Responses are checked for satisfaction of verifiable instruction constraints.",
      "order": 250,
      "coverageNote": "Candidate: public current requested-model values are conflicting or unavailable."
    },
    {
      "id": "audit-input-cost",
      "benchmarkId": "input-cost",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "Input price",
      "description": "Public uncached input price per million tokens.",
      "domain": [
        "pricing",
        "deployment"
      ],
      "organization": "Model providers",
      "benchmarkUrl": "https://platform.claude.com/docs/en/about-claude/pricing",
      "benchmarkVersion": "Public rate-card snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Input price per million tokens",
      "scoreType": "cost",
      "unit": "usd",
      "precision": 2,
      "direction": "lower",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-anthropic-pricing",
      "methodology": "Provider-published uncached input rate; long-context and cache tiers remain separate pricing records.",
      "order": 260,
      "coverageNote": "Candidate operational metric; not a benchmark evaluation."
    },
    {
      "id": "audit-median-latency",
      "benchmarkId": "median-latency",
      "auditState": "rejected",
      "categoryId": "deployment",
      "label": "Median latency (legacy generic label)",
      "description": "Legacy row with no defined workload, percentile, region, or response boundary.",
      "domain": [
        "latency",
        "deployment"
      ],
      "organization": "The Latent",
      "benchmarkUrl": "https://artificialanalysis.ai/methodology/performance-benchmarking",
      "benchmarkVersion": "Unpinned legacy label",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Median latency",
      "scoreType": "latency",
      "unit": "seconds",
      "precision": 2,
      "direction": "lower",
      "allowLimitedComparability": true,
      "scoreStatus": "archived",
      "citationId": "citation-aa-performance",
      "methodology": "Rejected because the legacy row does not define a reproducible latency workload.",
      "order": 270,
      "rejectionReason": "Replace only with explicit P50 TTFT or end-to-end workload definitions."
    },
    {
      "id": "audit-context-window",
      "benchmarkId": "context-window",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "Context window",
      "description": "Provider-published maximum context window.",
      "domain": [
        "context",
        "deployment"
      ],
      "organization": "Model providers",
      "benchmarkUrl": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "benchmarkVersion": "Provider documentation snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Maximum context window",
      "scoreType": "custom",
      "unit": "score",
      "precision": 0,
      "direction": "none",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-anthropic-opus-5-models",
      "methodology": "Provider-published maximum context value; not a benchmark and not ranked.",
      "order": 280,
      "coverageNote": "Candidate operational metadata; not a benchmark evaluation."
    },
    {
      "id": "audit-legacy-legal-workbench",
      "benchmarkId": "legal-workbench",
      "auditState": "rejected",
      "categoryId": "professional-work",
      "label": "Legal Workbench (legacy pseudo-benchmark)",
      "description": "Legacy conceptual label without a public benchmark definition.",
      "domain": [
        "legal reasoning"
      ],
      "organization": "Unverified",
      "benchmarkUrl": "https://arxiv.org/abs/2308.11432",
      "benchmarkVersion": "Not a public benchmark",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Undefined",
      "scoreType": "custom",
      "unit": "score",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "archived",
      "citationId": "citation-legalbench",
      "methodology": "This label was removed because a conceptual capability name cannot stand in for LegalBench or another concrete evaluation.",
      "order": 290,
      "rejectionReason": "No benchmark, native metric, or public result source under this name."
    },
    {
      "id": "audit-legacy-science-reasoning",
      "benchmarkId": "science-reasoning",
      "auditState": "rejected",
      "categoryId": "science",
      "label": "Science Reasoning (legacy pseudo-benchmark)",
      "description": "Legacy conceptual section label incorrectly represented as an evaluation.",
      "domain": [
        "science"
      ],
      "organization": "Unverified",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "benchmarkVersion": "Not a public benchmark",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Undefined",
      "scoreType": "custom",
      "unit": "score",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "archived",
      "citationId": "citation-aa-gpqa",
      "methodology": "The Science category remains; GPQA Diamond is the concrete verified row.",
      "order": 300,
      "rejectionReason": "Duplicate conceptual label; no independent benchmark identity."
    },
    {
      "id": "audit-legacy-document-reasoning",
      "benchmarkId": "document-reasoning",
      "auditState": "rejected",
      "categoryId": "multimodal",
      "label": "Document Reasoning (legacy pseudo-benchmark)",
      "description": "Legacy conceptual label incorrectly represented as an evaluation.",
      "domain": [
        "documents",
        "multimodal"
      ],
      "organization": "Unverified",
      "benchmarkUrl": "https://surgehq.ai/leaderboards/gdp-pdf",
      "benchmarkVersion": "Not a public benchmark",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Undefined",
      "scoreType": "custom",
      "unit": "score",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "archived",
      "citationId": "citation-gdp-pdf",
      "methodology": "GDP.pdf is tracked as a separate candidate with its own native metric and coverage status.",
      "order": 310,
      "rejectionReason": "No benchmark identity or native metric under the generic label."
    },
    {
      "id": "audit-legacy-long-form-factuality",
      "benchmarkId": "long-form-factuality",
      "auditState": "rejected",
      "categoryId": "factuality-calibration",
      "label": "Long-form factuality error (legacy pseudo-benchmark)",
      "description": "Legacy error label without a specific evaluation or native metric.",
      "domain": [
        "factuality"
      ],
      "organization": "Unverified",
      "benchmarkUrl": "https://arxiv.org/abs/2403.18802",
      "benchmarkVersion": "Not a public benchmark",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Undefined",
      "scoreType": "error",
      "unit": "percent",
      "precision": 1,
      "direction": "lower",
      "allowLimitedComparability": false,
      "scoreStatus": "archived",
      "citationId": "citation-longfact",
      "methodology": "LongFact + SAFE remains a candidate with its supported-fact metrics.",
      "order": 320,
      "rejectionReason": "Invented metric wording cannot be used as a native benchmark row."
    },
    {
      "id": "audit-legacy-simple-factuality",
      "benchmarkId": "simple-factuality",
      "auditState": "rejected",
      "categoryId": "factuality-calibration",
      "label": "Short factuality (legacy pseudo-benchmark)",
      "description": "Legacy conceptual label without an exact short-form factuality benchmark.",
      "domain": [
        "factuality"
      ],
      "organization": "Unverified",
      "benchmarkUrl": "https://arxiv.org/abs/2411.04368",
      "benchmarkVersion": "Not a public benchmark",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Undefined",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "archived",
      "citationId": "citation-simpleqa",
      "methodology": "SimpleQA remains a candidate with its own exact benchmark identity.",
      "order": 330,
      "rejectionReason": "Conceptual label cannot stand in for SimpleQA."
    },
    {
      "id": "audit-legacy-calibration-quality",
      "benchmarkId": "calibration-quality",
      "auditState": "rejected",
      "categoryId": "factuality-calibration",
      "label": "Calibration quality (legacy pseudo-benchmark)",
      "description": "Legacy conceptual label without a consistent public scalar.",
      "domain": [
        "calibration"
      ],
      "organization": "Unverified",
      "benchmarkUrl": "https://confidencebench.com/",
      "benchmarkVersion": "Not a public benchmark",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Undefined",
      "scoreType": "custom",
      "unit": "score",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "archived",
      "citationId": "citation-confidencebench",
      "methodology": "ConfidenceBench remains a candidate; AA-Omniscience Index is tracked separately as knowledge reliability.",
      "order": 340,
      "rejectionReason": "No consistent scalar named Calibration quality exists for the requested roster."
    },
    {
      "id": "audit-cursorbench-3-2",
      "benchmarkId": "cursorbench-3-2",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "CursorBench 3.2",
      "description": "IDE-native, multi-file coding-agent workflows evaluated in Cursor.",
      "domain": [
        "coding agents",
        "software engineering",
        "IDE workflows"
      ],
      "organization": "Cursor",
      "benchmarkUrl": "https://cursor.com/cursorbench",
      "benchmarkVersion": "CursorBench 3.2",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Task success rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "included",
      "citationId": "citation-cursorbench-32",
      "methodology": "Cursor evaluates ambiguous multi-file coding tasks in its productized agent workflow. The primary score is task success; cost, tokens, and steps are secondary metrics and are not folded into the score.",
      "order": 30,
      "coverageNote": "Vendor-run benchmark with partial public roster coverage; source configurations and missing cells are retained explicitly."
    },
    {
      "id": "audit-deepswe-1-1",
      "benchmarkId": "deepswe-1-1",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "DeepSWE v1.1",
      "description": "Long-horizon software-engineering agent evaluation on original repository tasks.",
      "domain": [
        "coding agents",
        "software engineering",
        "repositories"
      ],
      "organization": "Datacurve",
      "benchmarkUrl": "https://deepswe.datacurve.ai/",
      "benchmarkVersion": "DeepSWE v1.1",
      "benchmarkVersionDate": "2026-08-13T00:00:00.000Z",
      "metricName": "Pass@1",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "included",
      "citationId": "citation-deepswe-11",
      "methodology": "113 original tasks across 91 repositories are run with the mini-swe-agent harness and graded in clean isolated environments. Pass@1 is the native primary metric.",
      "order": 40,
      "coverageNote": "All seven target rows are sourced from the pinned DeepSWE v1.1 live artifact, including Gemini 3.1 Pro Preview High at 11.7257% Pass@1."
    },
    {
      "id": "audit-apex-agents",
      "benchmarkId": "apex-agents",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "APEX-Agents",
      "description": "Long-horizon cross-application professional-services agent evaluation.",
      "domain": [
        "agents",
        "professional work",
        "cross-application tasks"
      ],
      "organization": "Mercor",
      "benchmarkUrl": "https://www.mercor.com/apex/apex-agents-leaderboard/",
      "benchmarkVersion": "APEX-Agents public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Mean score",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-apex-agents",
      "methodology": "Agents complete 480 professional-services tasks across investment banking, consulting, and legal work. The supported public score is the mean rubric score; Pass@1 is tracked separately and is not substituted into this row.",
      "order": 50,
      "coverageNote": "The public leaderboard is configuration-specific and partial; no effort or model result is inferred from mean score.",
      "admissionStatus": "excluded",
      "admissionReason": "Excluded from scoring: more than two frontier cells are non-exact or unavailable; mean score and Pass@1 remain separate evidence."
    },
    {
      "id": "audit-apex-swe",
      "benchmarkId": "apex-swe",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "APEX-SWE",
      "description": "Real-world software-engineering work across integration and observability tasks.",
      "domain": [
        "software engineering",
        "agents",
        "integration",
        "observability"
      ],
      "organization": "Mercor and Cognition",
      "benchmarkUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/",
      "benchmarkVersion": "APEX-SWE public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass@1",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-apex-swe",
      "methodology": "Agents are evaluated on 200 held-out integration and observability cases with human-authored rubrics and tests. The leaderboard reports Pass@1.",
      "order": 60,
      "coverageNote": "Public leaderboard coverage is partial and effort-specific; missing model/configuration cells remain explicit.",
      "admissionStatus": "excluded",
      "admissionReason": "Excluded from scoring: more than two frontier cells are non-exact or unavailable."
    },
    {
      "id": "audit-aa-briefcase",
      "benchmarkId": "aa-briefcase",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "AA-Briefcase",
      "description": "Long-horizon professional knowledge-work evaluation producing real business deliverables.",
      "domain": [
        "professional work",
        "agents",
        "documents",
        "presentations"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/articles/aa-briefcase",
      "benchmarkVersion": "AA-Briefcase public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "AA-Briefcase Elo",
      "scoreType": "elo",
      "unit": "elo",
      "precision": 0,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-aa-briefcase",
      "methodology": "The benchmark combines rubric outcomes with analytical-quality and presentation-quality pairwise judgments into a separately defined overall Elo. The component metrics are retained as evidence, not treated as independent TLS benchmarks.",
      "order": 70,
      "coverageNote": "Private evaluation scenarios and rolling leaderboard updates mean the public configuration grid is partial; no composite value is inferred.",
      "admissionStatus": "excluded",
      "admissionReason": "Excluded from scoring: more than two frontier cells are non-exact or unavailable."
    },
    {
      "id": "audit-harvey-lab-aa",
      "benchmarkId": "harvey-lab-aa",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Harvey LAB-AA",
      "description": "Legal-agent evaluation on real-world legal deliverables across 24 practice areas.",
      "domain": [
        "legal",
        "agents",
        "professional work"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/evaluations/harvey-lab-aa",
      "benchmarkVersion": "Harvey LAB-AA public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Criterion pass rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-harvey-lab-aa",
      "methodology": "Agents produce legal deliverables on 120 private tasks. The primary stored metric is the share of atomic rubric criteria satisfied; the stricter all-pass rate remains a separate secondary metric.",
      "order": 80,
      "coverageNote": "The public leaderboard covers a partial current roster; criterion pass rate is not substituted with all-pass rate.",
      "admissionStatus": "excluded",
      "admissionReason": "Excluded from scoring: more than two frontier cells are non-exact or unavailable."
    },
    {
      "id": "audit-apex-agents-pass-at-1",
      "benchmarkId": "apex-agents-pass-at-1",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "APEX-Agents Pass@1 detail",
      "description": "Primary success metric for the APEX-Agents benchmark, tracked separately while the current result export is stabilized.",
      "domain": [
        "agents",
        "professional work"
      ],
      "organization": "Mercor",
      "benchmarkUrl": "https://www.mercor.com/apex/apex-agents-leaderboard/",
      "benchmarkVersion": "APEX-Agents public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass@1",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-apex-agents",
      "methodology": "Pass@1 requires one run to satisfy every task criterion. This candidate record prevents the primary metric from being confused with the currently more extractable mean score row.",
      "order": 85,
      "coverageNote": "Candidate until the full current configuration-level Pass@1 table is pinned; the supported APEX-Agents row uses the exact public mean-score metric instead."
    },
    {
      "id": "audit-terminal-bench-3",
      "benchmarkId": "terminal-bench-3",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "Terminal-Bench 3.0",
      "description": "Next-version terminal agent benchmark under active public leaderboard development.",
      "domain": [
        "agents",
        "terminal use",
        "coding"
      ],
      "organization": "Terminal-Bench",
      "benchmarkUrl": "https://www.tbench.ai/",
      "benchmarkVersion": "Terminal-Bench v3.0",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-terminal-bench",
      "methodology": "Candidate pending a stable canonical v3.0 release and configuration-level public leaderboard.",
      "order": 215,
      "coverageNote": "Candidate only; not admitted to the scored matrix or TLS until the version and source table stabilize."
    },
    {
      "id": "audit-humanitys-last-exam",
      "benchmarkId": "humanitys-last-exam",
      "auditState": "candidate",
      "categoryId": "science",
      "label": "Humanity's Last Exam",
      "description": "Broad expert-authored multimodal academic reasoning benchmark at the frontier of human knowledge.",
      "domain": [
        "general reasoning",
        "science",
        "mathematics",
        "humanities"
      ],
      "organization": "Center for AI Safety and collaborators",
      "benchmarkUrl": "https://lastexam.ai/",
      "benchmarkVersion": "HLE public leaderboard snapshot · 2,500-question release",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Accuracy (pass@1)",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-humanitys-last-exam",
      "methodology": "Humanity's Last Exam contains 2,500 expert-vetted questions across more than one hundred subjects. Public result sources use accuracy on a stated text-only or multimodal subset and may also report calibration error; accuracy is the candidate native capability metric.",
      "order": 350,
      "coverageNote": "Candidate: public leaderboards mix text-only subsets, multimodal subsets, and differing evaluation snapshots, so the requested configuration grid is not yet pinned."
    },
    {
      "id": "audit-aime-2026",
      "benchmarkId": "aime-2026",
      "auditState": "candidate",
      "categoryId": "mathematics",
      "label": "AIME 2026",
      "description": "Fresh competition-math evaluation using the 2026 AIME I and II problem sets.",
      "domain": [
        "mathematics",
        "competition reasoning"
      ],
      "organization": "American Invitational Mathematics Examination",
      "benchmarkUrl": "https://artofproblemsolving.com/wiki/index.php/AIME",
      "benchmarkVersion": "AIME 2026 I + II",
      "benchmarkVersionDate": "2026-02-11T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-aime-2026",
      "methodology": "The 2026 AIME I and II competitions provide 30 integer-answer problems. Accuracy is the share of problems solved exactly; the two contest sittings are kept together only when the source reports the combined 30-problem snapshot.",
      "order": 360,
      "coverageNote": "Candidate: public scores are predominantly provider-reported and often omit exact prompting, tool access, and reasoning-effort configuration."
    },
    {
      "id": "audit-chembench",
      "benchmarkId": "chembench",
      "auditState": "candidate",
      "categoryId": "science",
      "label": "ChemBench",
      "description": "Chemistry and materials-science benchmark covering knowledge, calculation, reasoning, and visual chemistry tasks.",
      "domain": [
        "chemistry",
        "materials science",
        "scientific reasoning"
      ],
      "organization": "LamaLab",
      "benchmarkUrl": "https://chembench.lamalab.org/",
      "benchmarkVersion": "ChemBench public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Overall accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-chembench",
      "methodology": "ChemBench reports performance across more than 2,700 curated chemistry questions and topic-specific tracks. The overall score is an accuracy aggregate; topic scores and refusal counts remain secondary evidence.",
      "order": 370,
      "coverageNote": "Candidate: the public leaderboard is real but current model rows, tool settings, and benchmark release pin are not yet normalized to the Latent configuration roster."
    },
    {
      "id": "audit-lab-bench",
      "benchmarkId": "lab-bench",
      "auditState": "candidate",
      "categoryId": "science",
      "label": "LAB-Bench",
      "description": "Biology research-agent benchmark covering literature, databases, figures, protocols, sequences, and cloning.",
      "domain": [
        "biology",
        "laboratory research",
        "scientific agents"
      ],
      "organization": "FutureHouse",
      "benchmarkUrl": "https://github.com/Future-House/LAB-Bench",
      "benchmarkVersion": "LAB-Bench public release",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Open-response mean accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-lab-bench",
      "methodology": "LAB-Bench spans eight broad categories and 30 subtasks, including LitQA2, DbQA, SuppQA, FigQA, TableQA, ProtocolQA, SeqQA, and Cloning Scenarios. Accuracy is reported by task and as an open-response mean; no tool-augmented result is substituted for the native no-tool evaluation.",
      "order": 380,
      "coverageNote": "Candidate: the benchmark and public data are real, but there is no maintainer-hosted current leaderboard covering the requested frontier configuration set."
    },
    {
      "id": "audit-discoverphysics",
      "benchmarkId": "discoverphysics",
      "auditState": "candidate",
      "categoryId": "physics",
      "label": "DiscoverPhysics",
      "description": "Open-ended scientific-discovery benchmark in simulated worlds with non-canonical physical laws.",
      "domain": [
        "physics",
        "scientific discovery",
        "agents"
      ],
      "organization": "DiscoverPhysics authors",
      "benchmarkUrl": "https://sampsonml.github.io/DiscoverPhysicsLeaderboard/",
      "benchmarkVersion": "DiscoverPhysics v1 public leaderboard snapshot",
      "benchmarkVersionDate": "2026-06-28T00:00:00.000Z",
      "metricName": "Pass@5",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-discoverphysics",
      "methodology": "Agents infer unfamiliar laws from simulated worlds. A world passes when normalized trajectory error and explanation quality both clear the published thresholds; Pass@k is estimated over sampled seeds and is kept distinct from explanation score and normalized MSE.",
      "order": 390,
      "coverageNote": "Candidate: the public board has a small research cohort and private worlds, so current frontier-model and effort-level coverage is not sufficient for TLS."
    },
    {
      "id": "audit-scicode",
      "benchmarkId": "scicode",
      "auditState": "verified",
      "categoryId": "coding-computation",
      "label": "SciCode",
      "description": "Scientist-curated scientific-programming benchmark spanning authentic laboratory problems.",
      "domain": [
        "scientific programming",
        "coding",
        "science"
      ],
      "organization": "SciCode authors",
      "benchmarkUrl": "https://scicode-bench.github.io/leaderboard/",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Main problem resolve rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-sci-code",
      "methodology": "The requested values are transcribed from the supplied comparison sources. Because those sources do not fully pin the underlying SciCode harness and reasoning settings, the row is displayed but excluded from aggregate scoring.",
      "order": 890,
      "coverageNote": "Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-terminal-bench-2-0",
      "benchmarkId": "terminal-bench-2-0",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "Terminal-Bench 2.0",
      "description": "Versioned terminal-agent benchmark covering software engineering, data science, security, and system tasks.",
      "domain": [
        "agents",
        "terminal use",
        "coding"
      ],
      "organization": "Terminal-Bench community",
      "benchmarkUrl": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
      "benchmarkVersion": "Terminal-Bench v2.0",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Task success rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-terminal-bench-2-0",
      "methodology": "Terminal-Bench 2.0 contains 89 terminal tasks with environment-specific tests and human-written solutions. Results are harness-specific and must remain separate from the later v2.1 correction and the proposed v3.0 candidate.",
      "order": 410,
      "coverageNote": "Candidate: v2.0 and v2.1 are both real but must not be mixed; current model rows also vary by agent harness."
    },
    {
      "id": "audit-charxiv",
      "benchmarkId": "charxiv",
      "auditState": "candidate",
      "categoryId": "multimodal",
      "label": "CharXiv",
      "description": "Human-curated scientific chart understanding benchmark with descriptive and cross-chart reasoning questions.",
      "domain": [
        "charts",
        "documents",
        "multimodal reasoning"
      ],
      "organization": "Princeton NLP",
      "benchmarkUrl": "https://charxiv.github.io/",
      "benchmarkVersion": "CharXiv v1.0",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Overall accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-charxiv",
      "methodology": "CharXiv uses 2,323 natural charts from scientific papers and separates descriptive from reasoning questions. Overall accuracy is reported alongside the two question-family scores; the source version and image/tool setting must be pinned before scoring.",
      "order": 420,
      "coverageNote": "Candidate: the canonical v1 leaderboard is real, but current frontier rows use heterogeneous multimodal prompts and do not yet provide the full effort-specific roster."
    },
    {
      "id": "audit-tau2-bench",
      "benchmarkId": "tau2-bench",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "τ²-bench",
      "description": "Tool-agent-user interaction benchmark for policy-constrained customer-service workflows.",
      "domain": [
        "agents",
        "tool use",
        "policy following"
      ],
      "organization": "Sierra Research",
      "benchmarkUrl": "https://taubench.com/",
      "benchmarkVersion": "τ²-bench public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass^k",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-tau2-bench",
      "methodology": "Agents converse with simulated users, call domain tools, and must produce a correct final database state while following policy. Pass^k measures repeated-task consistency and is kept distinct from pass@k and from the later τ³ extensions.",
      "order": 430,
      "coverageNote": "Candidate: the public leaderboard mixes domain, simulator, voice, and text configurations; a single comparable frontier text snapshot is not yet frozen."
    },
    {
      "id": "audit-bfcl-v4",
      "benchmarkId": "bfcl-v4",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "BFCL v4",
      "description": "Berkeley function-calling benchmark covering live, non-live, multi-turn, agentic, and hallucination tracks.",
      "domain": [
        "function calling",
        "tool use",
        "agents"
      ],
      "organization": "Berkeley Gorilla team",
      "benchmarkUrl": "https://gorilla.cs.berkeley.edu/leaderboard.html",
      "benchmarkVersion": "BFCL v4",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Overall score",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-bfcl-v4",
      "methodology": "BFCL v4 combines Agentic 40%, Multi-Turn 30%, Live 10%, Non-Live 10%, and Hallucination 10% components. The overall score is kept distinct from each component and requires the benchmark's exact tool schema and execution protocol.",
      "order": 440,
      "coverageNote": "Candidate: the live leaderboard contains heterogeneous model handlers and tool settings; the requested roster does not yet have a pinned, configuration-comparable export."
    },
    {
      "id": "audit-browsecomp",
      "benchmarkId": "browsecomp",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "BrowseComp",
      "description": "Persistent web-research benchmark for obscure, multi-hop information acquisition.",
      "domain": [
        "browsing",
        "research agents",
        "web search"
      ],
      "organization": "OpenAI",
      "benchmarkUrl": "https://openai.com/index/browsecomp",
      "benchmarkVersion": "BrowseComp public release",
      "benchmarkVersionDate": "2025-04-10T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-browsecomp",
      "methodology": "BrowseComp contains 1,266 difficult questions whose answers require persistent browsing and multi-hop synthesis. Accuracy is exact-answer correctness, but scores depend materially on search provider, budget, and agent scaffold.",
      "order": 450,
      "coverageNote": "Candidate: source results are system-level and search-configuration-specific rather than base-model-only, so the harness identity must be captured before TLS admission."
    },
    {
      "id": "audit-osworld-2",
      "benchmarkId": "osworld-2",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "OSWorld 2.0",
      "description": "Long-horizon desktop computer-use benchmark over realistic professional workflows.",
      "domain": [
        "computer use",
        "desktop agents",
        "long-horizon workflows"
      ],
      "organization": "xlang AI and collaborators",
      "benchmarkUrl": "https://osworld-v2.xlang.ai/",
      "benchmarkVersion": "OSWorld 2.0",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Binary completion rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-osworld-2",
      "methodology": "OSWorld 2.0 evaluates 108 workflows under a 500-step budget. Binary completion requires every checkpoint to be satisfied; partial checkpoint score is a separate secondary metric and is not substituted for binary completion.",
      "order": 460,
      "coverageNote": "Candidate: public results are agent/scaffold-specific and frequently report partial score without the corresponding binary-completion value."
    },
    {
      "id": "audit-online-mind2web",
      "benchmarkId": "online-mind2web",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "Online Mind2Web",
      "description": "Live-web actuation benchmark covering multi-step tasks across real websites.",
      "domain": [
        "web agents",
        "browser actuation",
        "online tasks"
      ],
      "organization": "Ohio State University NLP Group",
      "benchmarkUrl": "https://huggingface.co/spaces/osunlp/Online_Mind2Web_Leaderboard",
      "benchmarkVersion": "Online Mind2Web v2 public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "WebJudge success rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-online-mind2web",
      "methodology": "Online Mind2Web evaluates 300 tasks across 136 live websites with automatic WebJudge evaluation and a separate human-evaluation table. The automatic and human metrics are not interchangeable.",
      "order": 470,
      "coverageNote": "Candidate: the live-site environment, WebJudge version, and agent trajectory protocol must be pinned before comparing provider model rows."
    },
    {
      "id": "audit-androidworld",
      "benchmarkId": "androidworld",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "AndroidWorld",
      "description": "Dynamic Android computer-use environment with durable state-based task rewards.",
      "domain": [
        "mobile agents",
        "computer use",
        "Android"
      ],
      "organization": "Google Research",
      "benchmarkUrl": "https://google-research.github.io/android_world/",
      "benchmarkVersion": "AndroidWorld public release",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Mean success rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-androidworld",
      "methodology": "AndroidWorld dynamically instantiates 116 tasks across 20 apps and grades durable Android state rewards. Mean success rate is the native primary metric; screenshot, accessibility-tree, model, and reflection scaffold settings must be retained.",
      "order": 480,
      "coverageNote": "Candidate: most public rows are complete agent systems rather than isolated model configurations, and current requested-roster coverage is sparse."
    },
    {
      "id": "audit-faith-eval",
      "benchmarkId": "faith-eval",
      "auditState": "candidate",
      "categoryId": "factuality-calibration",
      "label": "FaithEval",
      "description": "Contextual-faithfulness benchmark for unanswerable, inconsistent, and counterfactual evidence.",
      "domain": [
        "grounding",
        "faithfulness",
        "RAG"
      ],
      "organization": "Salesforce AI Research",
      "benchmarkUrl": "https://github.com/SalesforceAIResearch/FaithEval",
      "benchmarkVersion": "FaithEval public release",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Combined accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-faith-eval",
      "methodology": "FaithEval evaluates whether model answers stay faithful to supplied context across unanswerable, inconsistent, and counterfactual tasks. The task-level accuracies are retained separately; the combined score is only used when the source defines the aggregation.",
      "order": 490,
      "coverageNote": "Candidate: real benchmark and evaluators exist, but current public model coverage and judge/version details are insufficient for a frozen frontier comparison."
    },
    {
      "id": "audit-healthbench-consensus",
      "benchmarkId": "healthbench-consensus",
      "auditState": "candidate",
      "categoryId": "professional-work",
      "label": "HealthBench Consensus",
      "description": "Physician-consensus subset of OpenAI HealthBench for realistic medical conversations.",
      "domain": [
        "health",
        "medical reasoning",
        "professional work"
      ],
      "organization": "OpenAI",
      "benchmarkUrl": "https://cdn.openai.com/pdf/bd7a39d5-9e9f-47b3-903c-8b847ca650c7/healthbench_paper.pdf",
      "benchmarkVersion": "HealthBench Consensus",
      "benchmarkVersionDate": "2025-05-12T00:00:00.000Z",
      "metricName": "Mean normalized rubric score",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-healthbench",
      "methodology": "HealthBench Consensus retains 3,671 conversations with physician-consensus criteria. Responses are scored against conversation-specific rubrics and aggregated as a mean normalized score; judge model and response-length conditions must be retained.",
      "order": 500,
      "coverageNote": "Candidate: public results are provider-run and judge-dependent, with incomplete coverage for the requested providers and effort variants."
    },
    {
      "id": "audit-healthbench-hard",
      "benchmarkId": "healthbench-hard",
      "auditState": "candidate",
      "categoryId": "professional-work",
      "label": "HealthBench Hard",
      "description": "Difficult 1,000-example subset of HealthBench focused on challenging medical conversations.",
      "domain": [
        "health",
        "medical reasoning",
        "professional work"
      ],
      "organization": "OpenAI",
      "benchmarkUrl": "https://cdn.openai.com/pdf/bd7a39d5-9e9f-47b3-903c-8b847ca650c7/healthbench_paper.pdf",
      "benchmarkVersion": "HealthBench Hard",
      "benchmarkVersionDate": "2025-05-12T00:00:00.000Z",
      "metricName": "Mean normalized rubric score",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-healthbench",
      "methodology": "HealthBench Hard is a 1,000-example difficulty-selected subset of HealthBench. Rubric criteria are normalized per conversation and averaged; length-adjusted third-party scores are not substituted for the native mean score.",
      "order": 510,
      "coverageNote": "Candidate: the benchmark is real but current public scores are mostly provider/system reports and do not establish a common configuration or judge snapshot."
    },
    {
      "id": "audit-legalbench",
      "benchmarkId": "legalbench",
      "auditState": "candidate",
      "categoryId": "professional-work",
      "label": "LegalBench",
      "description": "Multi-task legal reasoning benchmark spanning doctrinal, interpretive, and practical legal tasks.",
      "domain": [
        "legal reasoning",
        "professional work"
      ],
      "organization": "LegalBench collaborators",
      "benchmarkUrl": "https://arxiv.org/abs/2308.11432",
      "benchmarkVersion": "LegalBench public release",
      "benchmarkVersionDate": "2023-08-21T00:00:00.000Z",
      "metricName": "Task accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-legalbench",
      "methodology": "LegalBench contains 162 tasks across six legal-reasoning categories with task-specific evaluation functions. Aggregate accuracy is only meaningful when the exact task release and macro-averaging rule are pinned.",
      "order": 520,
      "coverageNote": "Candidate: the benchmark identity is established, but current comparable public results for the requested frontier roster and exact task aggregation are incomplete."
    },
    {
      "id": "audit-plawbench",
      "benchmarkId": "plawbench",
      "auditState": "candidate",
      "categoryId": "professional-work",
      "label": "PLawBench",
      "description": "Rubric-based benchmark for public legal consultation, case analysis, and legal document generation.",
      "domain": [
        "legal reasoning",
        "professional work",
        "document generation"
      ],
      "organization": "PLawBench authors",
      "benchmarkUrl": "https://aclanthology.org/2026.acl-long.458/",
      "benchmarkVersion": "PLawBench public release",
      "benchmarkVersionDate": "2026-07-01T00:00:00.000Z",
      "metricName": "Overall scoring rate",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-plawbench",
      "methodology": "PLawBench contains 850 questions across 13 practical legal scenarios and approximately 12,500 expert-designed rubric items. The overall scoring rate aggregates rubric dimensions across the three task categories and must retain the evaluator version.",
      "order": 530,
      "coverageNote": "Candidate: public paper and result tables exist, but the released frontier cohort is small and LLM-judge settings are not yet aligned to the Latent roster."
    },
    {
      "id": "audit-ifhierbench",
      "benchmarkId": "ifhierbench",
      "auditState": "candidate",
      "categoryId": "instruction-following",
      "label": "IFHierBench",
      "description": "Hierarchical instruction-following benchmark for nested content and structural constraints.",
      "domain": [
        "instruction following",
        "structured generation"
      ],
      "organization": "IFHierBench authors",
      "benchmarkUrl": "https://arxiv.org/html/2607.27912v1",
      "benchmarkVersion": "IFHierBench v1",
      "benchmarkVersionDate": "2026-07-31T00:00:00.000Z",
      "metricName": "Prompt-level accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": false,
      "scoreStatus": "displayed",
      "citationId": "citation-ifhierbench",
      "methodology": "IFHierBench contains 600 prompts across four constraint-tree depths and 35 constraints, each checked by deterministic code at every scope. Prompt-level accuracy is the primary metric; depth-level accuracy is secondary.",
      "order": 540,
      "coverageNote": "Candidate: the benchmark is newly published with a small evaluated cohort and no current public 67-configuration result table."
    },
    {
      "id": "audit-longbench-v2",
      "benchmarkId": "longbench-v2",
      "auditState": "candidate",
      "categoryId": "science",
      "label": "LongBench v2",
      "description": "Long-context understanding and reasoning benchmark over realistic multi-task documents.",
      "domain": [
        "long context",
        "document reasoning",
        "code understanding"
      ],
      "organization": "LongBench authors",
      "benchmarkUrl": "https://longbench2.github.io/",
      "benchmarkVersion": "LongBench v2",
      "benchmarkVersionDate": "2025-01-01T00:00:00.000Z",
      "metricName": "Compensated overall accuracy with CoT",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-longbench-v2",
      "methodology": "LongBench v2 contains 503 questions spanning six long-context task categories and contexts up to two million words. The leaderboard reports zero-shot and zero-shot-plus-CoT settings; the compensated accuracy convention counts invalid outputs at the random-choice baseline.",
      "order": 550,
      "coverageNote": "Candidate: CoT prompting, invalid-answer compensation, and context-length bins must be held constant; current public coverage predates the requested model roster."
    },
    {
      "id": "audit-helmet",
      "benchmarkId": "helmet",
      "auditState": "candidate",
      "categoryId": "science",
      "label": "HELMET",
      "description": "Application-centric long-context benchmark covering retrieval, RAG, summarization, reranking, and citation generation.",
      "domain": [
        "long context",
        "RAG",
        "document reasoning"
      ],
      "organization": "Princeton NLP",
      "benchmarkUrl": "https://princeton-nlp.github.io/HELMET/",
      "benchmarkVersion": "HELMET public release",
      "benchmarkVersionDate": "2024-10-03T00:00:00.000Z",
      "metricName": "Overall performance",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-helmet",
      "methodology": "HELMET covers seven application-centric long-context categories with controllable lengths up to 128K tokens. The overall score is an aggregate across heterogeneous task metrics and is retained only when the source's published aggregation is reproduced.",
      "order": 560,
      "coverageNote": "Candidate: public leaderboard coverage is broad but older and mixes task families, lengths, and model context settings."
    },
    {
      "id": "audit-ruler",
      "benchmarkId": "ruler",
      "auditState": "candidate",
      "categoryId": "science",
      "label": "RULER",
      "description": "Synthetic long-context benchmark covering retrieval, multi-hop tracing, aggregation, and question answering.",
      "domain": [
        "long context",
        "retrieval",
        "reasoning"
      ],
      "organization": "NVIDIA",
      "benchmarkUrl": "https://github.com/NVIDIA/RULER",
      "benchmarkVersion": "RULER public release",
      "benchmarkVersionDate": "2024-04-08T00:00:00.000Z",
      "metricName": "Weighted average accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-ruler",
      "methodology": "RULER generates controlled examples across 13 tasks and four categories at several context lengths. Effective-length and weighted-average scores are distinct; the exact length schedule must be pinned before comparison.",
      "order": 570,
      "coverageNote": "Candidate: synthetic task generation and context-length schedule make result snapshots non-comparable unless the exact release and configuration are frozen."
    },
    {
      "id": "audit-memory-agent-bench",
      "benchmarkId": "memory-agent-bench",
      "auditState": "candidate",
      "categoryId": "agents-tool-use",
      "label": "MemoryAgentBench",
      "description": "Incremental multi-turn benchmark for retrieval, test-time learning, long-range understanding, and conflict resolution.",
      "domain": [
        "agent memory",
        "long context",
        "multi-turn interaction"
      ],
      "organization": "HUST AI and collaborators",
      "benchmarkUrl": "https://github.com/HUST-AI-HYZ/MemoryAgentBench",
      "benchmarkVersion": "MemoryAgentBench public release",
      "benchmarkVersionDate": "2025-07-07T00:00:00.000Z",
      "metricName": "Competency macro-average accuracy",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-memory-agent-bench",
      "methodology": "MemoryAgentBench evaluates four competencies through incrementally delivered multi-turn interactions. Accurate retrieval, test-time learning, long-range understanding, conflict resolution, and task-specific metrics are reported separately; a macro-average is considered only if the source task weighting is reproduced.",
      "order": 580,
      "coverageNote": "Candidate: the benchmark is a framework with heterogeneous task metrics and no current live leaderboard covering the requested model/effort roster."
    },
    {
      "id": "audit-strongreject",
      "benchmarkId": "strongreject",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "StrongREJECT",
      "description": "Jailbreak-susceptibility benchmark measuring harmful compliance, specificity, and convincingness.",
      "domain": [
        "safety",
        "jailbreak robustness",
        "harmful compliance"
      ],
      "organization": "StrongREJECT authors",
      "benchmarkUrl": "https://arxiv.org/abs/2402.10260",
      "benchmarkVersion": "StrongREJECT public release",
      "benchmarkVersionDate": "2024-02-15T00:00:00.000Z",
      "metricName": "StrongREJECT score",
      "scoreType": "error",
      "unit": "score",
      "precision": 3,
      "direction": "lower",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-strongreject",
      "methodology": "The score combines non-refusal with specificity and convincingness: (1 − refusal) × (specificity + convincingness) / 2. Lower is safer. It belongs in the deployment panel and is not averaged into capability TLS.",
      "order": 590,
      "coverageNote": "Panel candidate: judge version, jailbreak transformation, and safety policy materially affect the result; no TLS capability weight is assigned."
    },
    {
      "id": "audit-agentharm",
      "benchmarkId": "agentharm",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "AgentHarm",
      "description": "Agent-safety benchmark for harmful multi-step tasks, refusal, and post-jailbreak capability.",
      "domain": [
        "agent safety",
        "harmful compliance",
        "tool use"
      ],
      "organization": "UK AI Safety Institute and collaborators",
      "benchmarkUrl": "https://huggingface.co/datasets/ai-safety-institute/AgentHarm",
      "benchmarkVersion": "AgentHarm public release",
      "benchmarkVersionDate": "2024-10-11T00:00:00.000Z",
      "metricName": "HarmScore",
      "scoreType": "error",
      "unit": "percent",
      "precision": 1,
      "direction": "lower",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-agentharm",
      "methodology": "AgentHarm evaluates 110 malicious base behaviors, augmented variants, refusal, and harmful task completion through agent trajectories. HarmScore, RefusalRate, and NonRefusalHarmScore are distinct; HarmScore is the panel's primary risk metric and lower is better.",
      "order": 600,
      "coverageNote": "Panel candidate: public release coverage is partial and results are agent/scaffold-specific; it is not a capability score."
    },
    {
      "id": "audit-agentdojo",
      "benchmarkId": "agentdojo",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "AgentDojo",
      "description": "Dynamic tool-use environment for prompt-injection robustness and benign task utility.",
      "domain": [
        "prompt injection",
        "agent safety",
        "tool use"
      ],
      "organization": "ETH Zurich SPY Lab",
      "benchmarkUrl": "https://agentdojo.spylab.ai/",
      "benchmarkVersion": "AgentDojo public release",
      "benchmarkVersionDate": "2024-06-19T00:00:00.000Z",
      "metricName": "Security success rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-agentdojo",
      "methodology": "AgentDojo evaluates 97 realistic tasks and 629 security test cases with benign utility, utility under attack, and security metrics. Security success is distinct from benign task completion and must be reported with the attack and defense configuration.",
      "order": 610,
      "coverageNote": "Panel candidate: dynamic tasks, attacks, defenses, and tool suites change the estimand; no single current frontier snapshot is yet comparable."
    },
    {
      "id": "audit-cybench",
      "benchmarkId": "cybench",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "Cybench",
      "description": "Cybersecurity agent benchmark over capture-the-flag tasks with guided and unguided execution modes.",
      "domain": [
        "cybersecurity",
        "agents",
        "tool use"
      ],
      "organization": "Cybench authors",
      "benchmarkUrl": "https://cybench.github.io/",
      "benchmarkVersion": "Cybench public release",
      "benchmarkVersionDate": "2024-08-13T00:00:00.000Z",
      "metricName": "Unguided percentage solved",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-cybench",
      "methodology": "Cybench contains 40 cybersecurity tasks and reports unguided success, subtask-guided success, subtask percentage solved, and most difficult task solved. Unguided success is the primary panel metric; guided metrics are secondary.",
      "order": 620,
      "coverageNote": "Panel candidate: public rows mix harnesses and subsets, and the task suite is not a general capability benchmark."
    },
    {
      "id": "audit-mask",
      "benchmarkId": "mask",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "MASK",
      "description": "Honesty benchmark testing whether pressured statements contradict a model's elicited beliefs.",
      "domain": [
        "honesty",
        "alignment",
        "reliability"
      ],
      "organization": "Center for AI Safety and collaborators",
      "benchmarkUrl": "https://www.mask-benchmark.ai/",
      "benchmarkVersion": "MASK public release",
      "benchmarkVersionDate": "2025-03-05T00:00:00.000Z",
      "metricName": "Honesty rate",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-mask",
      "methodology": "MASK first elicits a model's belief and then tests whether a pressure prompt induces a contradiction. Honesty is statement-belief agreement, while accuracy is belief-ground-truth agreement; the panel keeps them separate.",
      "order": 630,
      "coverageNote": "Panel candidate: pressure prompts, belief elicitation, and judge settings are part of the estimand, and current public results do not cover the requested effort roster."
    },
    {
      "id": "audit-marco-bench-mif",
      "benchmarkId": "marco-bench-mif",
      "auditState": "candidate",
      "categoryId": "instruction-following",
      "label": "Marco-Bench-MIF",
      "description": "Deeply localized multilingual instruction-following benchmark spanning 30 languages.",
      "domain": [
        "multilingual",
        "instruction following",
        "cultural reasoning"
      ],
      "organization": "AIDC-AI",
      "benchmarkUrl": "https://aclanthology.org/2025.acl-long.1172/",
      "benchmarkVersion": "Marco-Bench-MIF public release",
      "benchmarkVersionDate": "2025-07-01T00:00:00.000Z",
      "metricName": "Instruction-following accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-marco-bench-mif",
      "methodology": "Marco-Bench-MIF evaluates 541 localized instruction-response pairs per language across 30 languages and six language families. Scores must preserve language macro-averaging and cannot be treated as a simple English translation test.",
      "order": 640,
      "coverageNote": "Panel candidate: current public model results and language sampling are insufficient for a stable comparison across the requested roster."
    },
    {
      "id": "audit-global-mmlu",
      "benchmarkId": "global-mmlu",
      "auditState": "candidate",
      "categoryId": "science",
      "label": "Global-MMLU",
      "description": "Multilingual and culturally annotated extension of MMLU across 42 languages.",
      "domain": [
        "multilingual",
        "knowledge",
        "reasoning"
      ],
      "organization": "Cohere Labs and collaborators",
      "benchmarkUrl": "https://huggingface.co/datasets/CohereLabs/Global-MMLU",
      "benchmarkVersion": "Global-MMLU public release",
      "benchmarkVersionDate": "2025-01-01T00:00:00.000Z",
      "metricName": "Overall accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-global-mmlu",
      "methodology": "Global-MMLU spans 42 languages and separates culturally sensitive from culturally agnostic questions. Full, annotated, and Lite subsets are distinct releases; no cross-subset aggregate is inferred.",
      "order": 650,
      "coverageNote": "Panel candidate: public result tables are sparse and the full versus Lite subsets have materially different translation and sampling properties."
    },
    {
      "id": "audit-eq-bench-4",
      "benchmarkId": "eq-bench-4",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "EQ-Bench 4",
      "description": "Human-judged emotional and social intelligence evaluation using pairwise comparisons.",
      "domain": [
        "social intelligence",
        "writing quality",
        "interpersonal judgment"
      ],
      "organization": "EQ-Bench authors",
      "benchmarkUrl": "https://eqbench.com/",
      "benchmarkVersion": "EQ-Bench 4",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Elo rating",
      "scoreType": "elo",
      "unit": "elo",
      "precision": 0,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-eq-bench-4",
      "methodology": "EQ-Bench 4 uses blind pairwise transcript judgments across six ability dimensions and fits a soft Bradley-Terry model, normalized to fixed anchors with bootstrap intervals. Informational trait scores are not part of the headline Elo.",
      "order": 660,
      "coverageNote": "Panel candidate: judge panel, persona models, transcript generation, and tournament composition are part of the measurement and current model coverage is incomplete."
    },
    {
      "id": "audit-chatbot-arena",
      "benchmarkId": "chatbot-arena",
      "auditState": "candidate",
      "categoryId": "deployment",
      "label": "LMSYS Chatbot Arena",
      "description": "Live human pairwise-preference evaluation for general chat model quality.",
      "domain": [
        "human preference",
        "general chat",
        "external validation"
      ],
      "organization": "LMSYS and UC Berkeley collaborators",
      "benchmarkUrl": "https://www.lmsys.org/blog/2024-03-01-policy/",
      "benchmarkVersion": "Chatbot Arena public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Bradley-Terry rating",
      "scoreType": "elo",
      "unit": "elo",
      "precision": 0,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-chatbot-arena",
      "methodology": "Chatbot Arena collects anonymized human pairwise preferences and estimates model ratings with a Bradley-Terry-style procedure and uncertainty resampling. It is an external preference sanity check, not a capability benchmark and not a TLS input.",
      "order": 670,
      "coverageNote": "External-check candidate: the leaderboard is continuously refreshed, prompt mix is user-generated, and ratings are not treated as interchangeable with task benchmark scores."
    },
    {
      "id": "audit-code-migration",
      "benchmarkId": "code-migration",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "Code Migration",
      "description": "Agentic code-migration evaluation using hidden behavioral tests across language and framework migration tasks.",
      "domain": [
        "code migration",
        "software engineering",
        "coding"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/code-migration",
      "benchmarkVersion": "Vals Code Migration · CLI split · 2026-08-15 snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "CLI migration pass rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-code-migration",
      "methodology": "The Vals CLI split evaluates completed migrations in an offline sandbox with hidden behavioral tests and anti-cheat checks. The CLI split is retained because it exposes a complete seven-model table; it is not mixed with the page headline aggregate.",
      "order": 700,
      "coverageNote": "The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned.",
      "requiredScaffold": "Vals AI Code Migration harness",
      "admissionStatus": "excluded",
      "admissionReason": "Imported for transparent comparison only. The captured source does not establish the required frontier effort and/or has more than two missing target cells."
    },
    {
      "id": "audit-excel-modeling-benchmark",
      "benchmarkId": "excel-modeling-benchmark",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Excel Modeling Benchmark",
      "description": "Financial spreadsheet construction benchmark covering investment banking and private-equity modeling tasks.",
      "domain": [
        "financial modeling",
        "spreadsheets",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/emb",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Dataroom Summaries accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-excel-modeling",
      "methodology": "The benchmark evaluates agents building Excel workbooks from realistic financial prompts and source files. This record uses the published Scratch-mode Dataroom Summaries slice, a native category metric, rather than deriving a synthetic average from the page-level headline and category tables.",
      "order": 710,
      "coverageNote": "The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score.",
      "requiredScaffold": "Vals AI Excel Modeling harness",
      "admissionStatus": "excluded",
      "admissionReason": "Imported for transparent comparison only. The captured source does not establish the required frontier effort and/or has more than two missing target cells."
    },
    {
      "id": "audit-legal-research-bench",
      "benchmarkId": "legal-research-bench",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Legal Research Bench",
      "description": "Long-horizon legal research evaluation using a shared browser and document-research tool harness.",
      "domain": [
        "legal research",
        "professional work",
        "agents"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/legal_research",
      "benchmarkVersion": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "All-pass accuracy · Health practice-area slice",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-legal-research",
      "methodology": "The Vals Legal Research Bench runs models through a common research environment and scores all-pass accuracy by area of law. This record uses the published Health practice-area slice because it is the stable row-level metric exposed for all seven target families; it does not infer the overall benchmark score from area values.",
      "order": 720,
      "coverageNote": "The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only.",
      "requiredScaffold": "Vals AI Legal Research harness",
      "admissionStatus": "excluded",
      "admissionReason": "Imported for transparent comparison only. The captured source does not establish the required frontier effort and/or has more than two missing target cells."
    },
    {
      "id": "audit-finance-agent-v2",
      "benchmarkId": "finance-agent-v2",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Finance Agent v2 — Vals",
      "description": "Tool-mediated financial analyst evaluation covering filings, comparables, calculations, and modeling.",
      "domain": [
        "finance",
        "professional work",
        "agents"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/fabv2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-finance-agent-v2",
      "methodology": "The requested values are reported under one named comparison family. Vals AI Finance Agent v2 harness Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 950,
      "coverageNote": "Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank.",
      "requiredScaffold": "Vals AI Finance Agent v2 harness",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-harvey-lab-vals",
      "benchmarkId": "harvey-lab-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Harvey Legal Agent Benchmark · Vals run",
      "description": "Legal-agent benchmark run on a shared Vals tool harness, kept separate from the Artificial Analysis Harvey LAB-AA record.",
      "domain": [
        "legal agents",
        "professional work",
        "agents"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/hlab",
      "benchmarkVersion": "Vals Harvey LAB · public leaderboard snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-harvey-lab",
      "methodology": "The Vals Harvey LAB run evaluates legal-agent trajectories with a common tool environment. It is a distinct source and harness from Harvey LAB-AA. Fable 5 fallback routing is preserved in the row note and is not treated as a pure Fable measurement.",
      "order": 740,
      "coverageNote": "The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted.",
      "requiredScaffold": "Vals AI Harvey LAB harness",
      "admissionStatus": "excluded",
      "admissionReason": "Imported for transparent comparison only. The captured source does not establish the required frontier effort and/or has more than two missing target cells."
    },
    {
      "id": "audit-vibe-code-bench",
      "benchmarkId": "vibe-code-bench",
      "auditState": "verified",
      "categoryId": "coding-computation",
      "label": "Vibe Code Bench v1.1",
      "description": "End-to-end web application construction benchmark with browser, database, and long-running agent evaluation.",
      "domain": [
        "coding",
        "web development",
        "agents"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/vibe-code",
      "benchmarkVersion": "Vals Vibe Code Bench v1.1 · public leaderboard snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Application quality score",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-vibe-code",
      "methodology": "Vibe Code Bench evaluates complete web applications in an OpenHands-style environment with browser and database tools. The benchmark version is kept separate from other Vals coding runs because the scaffold and evaluator materially affect the estimand.",
      "order": 750,
      "coverageNote": "The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks.",
      "requiredScaffold": "Vals AI Vibe Code Bench v1.1 harness",
      "admissionStatus": "excluded",
      "admissionReason": "Imported for transparent comparison only. The captured source does not establish the required frontier effort and/or has more than two missing target cells."
    },
    {
      "id": "audit-mmlu-pro",
      "benchmarkId": "mmlu-pro",
      "auditState": "verified",
      "categoryId": "science",
      "label": "MMLU Pro — Vals",
      "description": "Broad knowledge and reasoning evaluation using the Vals five-shot protocol.",
      "domain": [
        "knowledge",
        "reasoning",
        "science"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/mmlu_pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-mmlu-pro",
      "methodology": "The requested values are reported under one named comparison family. Vals AI five-shot MMLU-Pro harness Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1000,
      "coverageNote": "Requested max-configuration values are retained under Vals AI five-shot MMLU-Pro harness; absent models remain blank.",
      "requiredScaffold": "Vals AI five-shot MMLU-Pro harness",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-programbench",
      "benchmarkId": "programbench",
      "auditState": "verified",
      "categoryId": "coding-computation",
      "label": "ProgramBench · Vals run",
      "description": "Program synthesis and software construction benchmark with execution-based evaluation.",
      "domain": [
        "program synthesis",
        "coding",
        "software engineering"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/programbench",
      "benchmarkVersion": "Vals ProgramBench · public leaderboard snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Program success rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-programbench",
      "methodology": "ProgramBench evaluates generated programs with execution-based checks. The source page exposes a small number of frontier rows and includes fallback behavior for some systems; those caveats are retained rather than converted to a clean model-only score.",
      "order": 800,
      "coverageNote": "Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark.",
      "requiredScaffold": "Vals AI ProgramBench harness",
      "admissionStatus": "excluded",
      "admissionReason": "Imported for transparent comparison only. The captured source does not establish the required frontier effort and/or has more than two missing target cells."
    },
    {
      "id": "audit-apex-swe-integration",
      "benchmarkId": "apex-swe-integration",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "APEX-SWE · Integration",
      "description": "Software-integration engineering track from the APEX-SWE benchmark, kept separate from the aggregate leaderboard.",
      "domain": [
        "software engineering",
        "agents",
        "coding"
      ],
      "organization": "Mercor",
      "benchmarkUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/integration-swe/",
      "benchmarkVersion": "APEX-SWE Integration track · public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass@1",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-apex-swe-integration",
      "methodology": "The Integration track contains 100 held-out software-engineering cases and reports Pass@1. The public track rows identify the requested Max/High configurations for five frontier models; Sol XHigh and Grok High rows are not substituted for the requested Sol Max and Grok XHigh cells.",
      "order": 820,
      "coverageNote": "Track-specific APEX-SWE evidence reaches five exact target configurations. The public pages identify the harness as Terminus-2, while the existing aggregate record uses Mercor Archipelago; the track rows remain excluded until that scaffold discrepancy is reconciled.",
      "requiredScaffold": "Terminus-2",
      "admissionStatus": "excluded",
      "admissionReason": "Track-specific rows pass the numerical 5/7 gate, but the public Terminus-2 scaffold must be reconciled with the local aggregate Archipelago record before scoring."
    },
    {
      "id": "audit-apex-swe-observability",
      "benchmarkId": "apex-swe-observability",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "APEX-SWE · Observability",
      "description": "Software-observability engineering track from the APEX-SWE benchmark, kept separate from the aggregate leaderboard.",
      "domain": [
        "software engineering",
        "agents",
        "coding"
      ],
      "organization": "Mercor",
      "benchmarkUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/observability-swe/",
      "benchmarkVersion": "APEX-SWE Observability track · public leaderboard snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass@1",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-apex-swe-observability",
      "methodology": "The Observability track contains 100 held-out software-engineering cases and reports Pass@1. The public track rows identify the requested Max/High configurations for five frontier models; Sol XHigh and Grok High rows are not substituted for the requested Sol Max and Grok XHigh cells.",
      "order": 830,
      "coverageNote": "Track-specific APEX-SWE evidence reaches five exact target configurations. The public pages identify the harness as Terminus-2, while the existing aggregate record uses Mercor Archipelago; the track rows remain excluded until that scaffold discrepancy is reconciled.",
      "requiredScaffold": "Terminus-2",
      "admissionStatus": "excluded",
      "admissionReason": "Track-specific rows pass the numerical 5/7 gate, but the public Terminus-2 scaffold must be reconciled with the local aggregate Archipelago record before scoring."
    },
    {
      "id": "audit-matharena-arxivmath-2026-06",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "auditState": "verified",
      "categoryId": "mathematics",
      "label": "MathArena — ArXivMath Jun 2026",
      "description": "June 2026 MathArena accuracy on recent arXiv mathematics problems.",
      "domain": [
        "mathematics",
        "research reasoning"
      ],
      "organization": "MathArena",
      "benchmarkUrl": "https://matharena.ai/",
      "benchmarkVersion": "MathArena ArXivMath · 06/2026",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-matharena",
      "methodology": "Accuracy on the same 06/2026 ArXivMath competition is read from each evaluator-hosted max-configuration model page. Confidence intervals are retained on each observation.",
      "order": 860,
      "coverageNote": "The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result.",
      "requiredScaffold": "MathArena June 2026 competition harness",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-matharena-brokenarxiv-2026-06",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "auditState": "verified",
      "categoryId": "mathematics",
      "label": "MathArena — BrokenArXiv Jun 2026",
      "description": "June 2026 MathArena accuracy on deliberately malformed arXiv math problems.",
      "domain": [
        "mathematics",
        "robust reasoning"
      ],
      "organization": "MathArena",
      "benchmarkUrl": "https://matharena.ai/",
      "benchmarkVersion": "MathArena BrokenArXiv · 06/2026",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-matharena",
      "methodology": "Accuracy on the same 06/2026 BrokenArXiv competition is read from each evaluator-hosted max-configuration model page. Confidence intervals are retained on each observation.",
      "order": 870,
      "coverageNote": "The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing.",
      "requiredScaffold": "MathArena June 2026 competition harness",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-matharena-arxivlean-2026-06",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "auditState": "verified",
      "categoryId": "mathematics",
      "label": "MathArena — ArXivLean Jun 2026",
      "description": "June 2026 MathArena accuracy on formalized arXiv mathematics problems.",
      "domain": [
        "mathematics",
        "formal reasoning"
      ],
      "organization": "MathArena",
      "benchmarkUrl": "https://matharena.ai/",
      "benchmarkVersion": "MathArena ArXivLean · 06/2026",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-matharena",
      "methodology": "Accuracy on the same 06/2026 ArXivLean competition is read from each evaluator-hosted max-configuration model page. Confidence intervals are retained on each observation.",
      "order": 880,
      "coverageNote": "The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing.",
      "requiredScaffold": "MathArena June 2026 competition harness",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-frontiercode-1-1-main",
      "benchmarkId": "frontiercode-1-1-main",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "FrontierCode v1.1 Main",
      "description": "Main-set FrontierCode v1.1 public leaderboard retained separately from the Extended subset.",
      "domain": [
        "coding",
        "software engineering"
      ],
      "organization": "Cognition",
      "benchmarkUrl": "https://cognition.ai/blog/frontier-code",
      "benchmarkVersion": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Main score",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-frontiercode-11",
      "methodology": "Scores use the FrontierCode v1.1 Main subset and are kept separate from Extended results. Model-specific effort and harness disclosures are retained; the row remains display-only because the public evidence is assembled across source records.",
      "order": 900,
      "coverageNote": "Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-matharena-arxivmath-2026-05",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "auditState": "verified",
      "categoryId": "mathematics",
      "label": "MathArena — ArXivMath May 2026",
      "description": "May 2026 MathArena accuracy on recent arXiv mathematics problems.",
      "domain": [
        "mathematics",
        "research reasoning"
      ],
      "organization": "MathArena",
      "benchmarkUrl": "https://matharena.ai/",
      "benchmarkVersion": "MathArena ArXivMath · 05/2026",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-matharena",
      "methodology": "Accuracy on the 05/2026 ArXivMath competition is read from evaluator-hosted Fable 5 max and Kimi K3 model pages. June results for GPT-5.6 Sol and Opus 5 are not inserted into this May row.",
      "order": 850,
      "coverageNote": "Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here.",
      "requiredScaffold": "MathArena May 2026 competition harness",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-sage-vals",
      "benchmarkId": "sage-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "SAGE — Vals",
      "description": "Student-assessment benchmark evaluated under the Vals comparison family.",
      "domain": [
        "education",
        "assessment",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "SAGE · Vals snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Score",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI max-compute evaluation Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 920,
      "coverageNote": "Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank.",
      "requiredScaffold": "Vals AI max-compute evaluation",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-corpfin-v2-vals",
      "benchmarkId": "corpfin-v2-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "CorpFin v2 — Vals",
      "description": "Long-credit-agreement comprehension benchmark across three context tasks.",
      "domain": [
        "finance",
        "long context",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://www.vals.ai/benchmarks/corp_fin_v2",
      "benchmarkVersion": "CorpFin v2 · Vals archived snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-corpfin-v2",
      "methodology": "The requested values are reported under one named comparison family. Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 930,
      "coverageNote": "Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank.",
      "requiredScaffold": "Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-mortgage-tax-vals",
      "benchmarkId": "mortgage-tax-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "MortgageTax — Vals",
      "description": "Vals mortgage and tax reasoning benchmark.",
      "domain": [
        "finance",
        "tax",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "MortgageTax · Vals snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI max-compute evaluation Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 940,
      "coverageNote": "Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank.",
      "requiredScaffold": "Vals AI max-compute evaluation",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-excel-modeling-benchmark-vals-overall",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Excel Modeling Benchmark — Vals",
      "description": "Overall Vals financial spreadsheet construction benchmark score.",
      "domain": [
        "financial modeling",
        "spreadsheets",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/emb",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-excel-modeling",
      "methodology": "The requested values are reported under one named comparison family. Vals AI Excel Modeling overall evaluation Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 960,
      "coverageNote": "Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank.",
      "requiredScaffold": "Vals AI Excel Modeling overall evaluation",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-taxeval-v2-vals",
      "benchmarkId": "taxeval-v2-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "TaxEval v2 — Vals",
      "description": "Versioned tax-domain evaluation from Vals AI.",
      "domain": [
        "tax",
        "finance",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI max-compute evaluation Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 970,
      "coverageNote": "Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank.",
      "requiredScaffold": "Vals AI max-compute evaluation",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-medcode-vals",
      "benchmarkId": "medcode-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "MedCode — Vals",
      "description": "Medical coding evaluation from the shared Vals evaluation family.",
      "domain": [
        "healthcare",
        "medical coding",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI health evaluation Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 980,
      "coverageNote": "Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank.",
      "requiredScaffold": "Vals AI health evaluation",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-medscribe-vals",
      "benchmarkId": "medscribe-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "MedScribe — Vals",
      "description": "Medical scribing evaluation from the shared Vals evaluation family.",
      "domain": [
        "healthcare",
        "medical scribing",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI health evaluation Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 990,
      "coverageNote": "Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank.",
      "requiredScaffold": "Vals AI health evaluation",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-vals-index",
      "benchmarkId": "vals-index",
      "auditState": "verified",
      "categoryId": "overview",
      "label": "Vals Index",
      "description": "Vals AI composite index across its public evaluation suite.",
      "domain": [
        "composite",
        "general capability"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Composite score",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI index methodology Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1010,
      "coverageNote": "Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank.",
      "requiredScaffold": "Vals AI index methodology",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-vals-multimodal-index",
      "benchmarkId": "vals-multimodal-index",
      "auditState": "verified",
      "categoryId": "multimodal",
      "label": "Vals Multimodal Index",
      "description": "Vals AI composite index over multimodal evaluations.",
      "domain": [
        "composite",
        "multimodal"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "Vals Multimodal Index snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Composite score",
      "scoreType": "custom",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI multimodal index methodology Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1020,
      "coverageNote": "Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank.",
      "requiredScaffold": "Vals AI multimodal index methodology",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-legal-research-bench-vals-overall",
      "benchmarkId": "legal-research-bench-vals-overall",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Legal Research Bench — Vals",
      "description": "Overall Vals long-horizon legal research evaluation.",
      "domain": [
        "legal research",
        "professional work",
        "agents"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://vals.ai/benchmarks/legal_research",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-vals-legal-research",
      "methodology": "The requested values are reported under one named comparison family. Vals AI Legal Research overall harness Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1030,
      "coverageNote": "Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank.",
      "requiredScaffold": "Vals AI Legal Research overall harness",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-legalbench-vals",
      "benchmarkId": "legalbench-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "LegalBench — Vals",
      "description": "Legal reasoning benchmark evaluated by Vals AI.",
      "domain": [
        "legal reasoning",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI LegalBench evaluation Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1040,
      "coverageNote": "Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank.",
      "requiredScaffold": "Vals AI LegalBench evaluation",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-public-benefits-bench-vals",
      "benchmarkId": "public-benefits-bench-vals",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Public Benefits Bench — Vals",
      "description": "Public-benefits eligibility and reasoning benchmark from Vals AI.",
      "domain": [
        "public benefits",
        "legal reasoning",
        "professional work"
      ],
      "organization": "Vals AI",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Vals AI public-benefits evaluation Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1050,
      "coverageNote": "Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank.",
      "requiredScaffold": "Vals AI public-benefits evaluation",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-aiiq-composite-iq",
      "benchmarkId": "aiiq-composite-iq",
      "auditState": "verified",
      "categoryId": "overview",
      "label": "AIIQ Composite IQ",
      "description": "IQ-like composite estimate across multiple reasoning dimensions.",
      "domain": [
        "composite",
        "reasoning"
      ],
      "organization": "AIIQ",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "AIIQ Composite IQ snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Composite IQ",
      "scoreType": "custom",
      "unit": "score",
      "precision": 0,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. AIIQ composite snapshot Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1060,
      "coverageNote": "Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank.",
      "requiredScaffold": "AIIQ composite snapshot",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-design-arena-elo",
      "benchmarkId": "design-arena-elo",
      "auditState": "verified",
      "categoryId": "overview",
      "label": "Design Arena (Elo)",
      "description": "Crowdsourced pairwise arena for AI-generated design outputs.",
      "domain": [
        "design",
        "human preference",
        "arena"
      ],
      "organization": "Design Arena",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "Design Arena rating snapshot · 2026-08-11",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Elo rating",
      "scoreType": "elo",
      "unit": "elo",
      "precision": 0,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are reported under one named comparison family. Design Arena 2026-08-11 rating snapshot Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1070,
      "coverageNote": "Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank.",
      "requiredScaffold": "Design Arena 2026-08-11 rating snapshot",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-frontier-bench-v0-1-anthropic-h2h",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "Frontier-Bench v0.1 — Anthropic H2H",
      "description": "Agentic terminal coding evaluation from Anthropic’s Opus 5 comparison.",
      "domain": [
        "coding",
        "agents",
        "terminal use"
      ],
      "organization": "Anthropic",
      "benchmarkUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "benchmarkVersion": "Frontier-Bench v0.1 · Anthropic Opus 5 head-to-head",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "methodology": "The requested values are reported under one named comparison family. Anthropic Opus 5 published head-to-head setup Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1080,
      "coverageNote": "Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank.",
      "requiredScaffold": "Anthropic Opus 5 published head-to-head setup",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-osworld-2-anthropic-h2h",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "OSWorld 2.0 — Anthropic H2H",
      "description": "Computer-use task success from Anthropic’s Opus 5 comparison.",
      "domain": [
        "computer use",
        "agents"
      ],
      "organization": "Anthropic",
      "benchmarkUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "benchmarkVersion": "OSWorld 2.0 · Anthropic Opus 5 head-to-head",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "methodology": "The requested values are reported under one named comparison family. Anthropic Opus 5 published head-to-head setup Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1090,
      "coverageNote": "Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank.",
      "requiredScaffold": "Anthropic Opus 5 published head-to-head setup",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-automationbench-anthropic-h2h",
      "benchmarkId": "automationbench-anthropic-h2h",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "AutomationBench — Anthropic H2H",
      "description": "Business-workflow automation benchmark from Anthropic’s Opus 5 comparison.",
      "domain": [
        "automation",
        "agents",
        "professional work"
      ],
      "organization": "Anthropic",
      "benchmarkUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "benchmarkVersion": "AutomationBench · Anthropic Opus 5 head-to-head",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "methodology": "The requested values are reported under one named comparison family. Anthropic Opus 5 published head-to-head setup Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1100,
      "coverageNote": "Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank.",
      "requiredScaffold": "Anthropic Opus 5 published head-to-head setup",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-harvey-legal-agent-held-out-h2h",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "Harvey Legal Agent Benchmark — held-out",
      "description": "Held-out legal-agent benchmark from Anthropic’s Opus 5 comparison.",
      "domain": [
        "legal agents",
        "professional work"
      ],
      "organization": "Anthropic",
      "benchmarkUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "benchmarkVersion": "Legal Agent Benchmark held-out · Anthropic Opus 5 head-to-head",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "methodology": "The requested values are reported under one named comparison family. Anthropic Opus 5 published held-out legal-agent setup Distinct evaluator versions and similarly named benchmarks remain separate.",
      "order": 1110,
      "coverageNote": "Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank.",
      "requiredScaffold": "Anthropic Opus 5 published held-out legal-agent setup",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-mcp-atlas-cross-source",
      "benchmarkId": "mcp-atlas-cross-source",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "MCP Atlas †",
      "description": "Tool-orchestration benchmark assembled from public model profiles.",
      "domain": [
        "tool use",
        "agents"
      ],
      "organization": "MCP Atlas",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "MCP Atlas · cross-source snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are retained as an informative cross-source comparison. Version-matched public model-profile results Public provenance does not establish one fully controlled harness, so the row is marked with a dagger and excluded from aggregate scoring.",
      "order": 1120,
      "coverageNote": "Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-code-migration-cross-source",
      "benchmarkId": "code-migration-cross-source",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "Code Migration †",
      "description": "Code-migration comparison assembled from public model profiles.",
      "domain": [
        "code migration",
        "software engineering"
      ],
      "organization": "BenchmarkList",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "Code Migration · cross-source snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are retained as an informative cross-source comparison. Version-matched public model-profile results Public provenance does not establish one fully controlled harness, so the row is marked with a dagger and excluded from aggregate scoring.",
      "order": 1130,
      "coverageNote": "Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-skillsbench-cross-source",
      "benchmarkId": "skillsbench-cross-source",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "SkillsBench †",
      "description": "SkillsBench comparison assembled from public model profiles.",
      "domain": [
        "skills",
        "agents",
        "professional work"
      ],
      "organization": "SkillsBench",
      "benchmarkUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "benchmarkVersion": "SkillsBench · cross-source snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Score",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are retained as an informative cross-source comparison. Version-matched public model-profile results Public provenance does not establish one fully controlled harness, so the row is marked with a dagger and excluded from aggregate scoring.",
      "order": 1140,
      "coverageNote": "Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-officeqa-pro-cross-source",
      "benchmarkId": "officeqa-pro-cross-source",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "OfficeQA Pro †",
      "description": "Office productivity question-answering benchmark from public model reports.",
      "domain": [
        "office productivity",
        "professional work"
      ],
      "organization": "OfficeQA Pro",
      "benchmarkUrl": "https://benchmarklist.com/benchmarks/officeqa_pro/",
      "benchmarkVersion": "OfficeQA Pro · cross-source snapshot · 2026-08-15",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-fable-5",
      "methodology": "The requested values are retained as an informative cross-source comparison. Mixed launch-post and system-card results Public provenance does not establish one fully controlled harness, so the row is marked with a dagger and excluded from aggregate scoring.",
      "order": 1150,
      "coverageNote": "Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-frontierswe",
      "benchmarkId": "frontierswe",
      "auditState": "verified",
      "categoryId": "software-engineering",
      "label": "FrontierSWE",
      "description": "FrontierSWE software-engineering agent evaluation.",
      "domain": [
        "software engineering",
        "coding agents"
      ],
      "organization": "BenchmarkList",
      "benchmarkUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "benchmarkVersion": "FrontierSWE · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Pass rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-kimi-k3",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 180,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-jobbench",
      "benchmarkId": "jobbench",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "JobBench",
      "description": "JobBench professional-work agent evaluation.",
      "domain": [
        "professional work",
        "agents"
      ],
      "organization": "BenchmarkList",
      "benchmarkUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "benchmarkVersion": "JobBench · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-kimi-k3",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 190,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-babyvision",
      "benchmarkId": "babyvision",
      "auditState": "verified",
      "categoryId": "multimodal",
      "label": "BabyVision (with CI)",
      "description": "BabyVision visual reasoning evaluation.",
      "domain": [
        "multimodal",
        "vision"
      ],
      "organization": "BenchmarkList",
      "benchmarkUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "benchmarkVersion": "BabyVision (with CI) · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-kimi-k3",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 200,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-perceptionbench",
      "benchmarkId": "perceptionbench",
      "auditState": "verified",
      "categoryId": "multimodal",
      "label": "PerceptionBench",
      "description": "PerceptionBench visual perception evaluation.",
      "domain": [
        "multimodal",
        "vision"
      ],
      "organization": "BenchmarkList",
      "benchmarkUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "benchmarkVersion": "PerceptionBench · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-kimi-k3",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 220,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-osworld-verified",
      "benchmarkId": "osworld-verified",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "OSWorld-Verified",
      "description": "Computer-use task success assembled from documented public model-agent runs on OSWorld-Verified.",
      "domain": [
        "computer use",
        "agents"
      ],
      "organization": "OSWorld authors",
      "benchmarkUrl": "https://os-world.github.io/",
      "benchmarkVersion": "OSWorld-Verified · cross-source snapshot · 2026-08-16",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-ouroboros-osworld-opus-5",
      "methodology": "Values are retained as cross-source model-agent evidence. The Opus 5 run publishes every task score and manifest but uses Ouroboros v6.87.0 rather than a common scaffold.",
      "order": 1095,
      "coverageNote": "Only directly documented OSWorld-Verified values are retained; OSWorld 2.0 and other task-set variants are not substituted.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-toolathlon-verified",
      "benchmarkId": "toolathlon-verified",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "Toolathlon-Verified",
      "description": "Toolathlon-Verified tool-use evaluation.",
      "domain": [
        "tool use",
        "agents"
      ],
      "organization": "BenchmarkList",
      "benchmarkUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "benchmarkVersion": "Toolathlon-Verified · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-kimi-k3",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 260,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-aa-lcr",
      "benchmarkId": "aa-lcr",
      "auditState": "verified",
      "categoryId": "factuality-calibration",
      "label": "AA-LCR",
      "description": "Artificial Analysis long-context retrieval evaluation.",
      "domain": [
        "long context",
        "retrieval"
      ],
      "organization": "Artificial Analysis",
      "benchmarkUrl": "https://artificialanalysis.ai/",
      "benchmarkVersion": "AA-LCR · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-kimi-k3",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 320,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-livebench",
      "benchmarkId": "livebench",
      "auditState": "verified",
      "categoryId": "overview",
      "label": "LiveBench",
      "description": "LiveBench contamination-resistant reasoning evaluation.",
      "domain": [
        "reasoning",
        "general capability"
      ],
      "organization": "LiveBench authors",
      "benchmarkUrl": "https://livebench.ai/",
      "benchmarkVersion": "LiveBench · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-kimi-k3",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 350,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-creative-writing-v3",
      "benchmarkId": "creative-writing-v3",
      "auditState": "verified",
      "categoryId": "overview",
      "label": "Creative Writing v3",
      "description": "EQ-Bench Creative Writing v3 Elo evaluation.",
      "domain": [
        "creative writing",
        "language"
      ],
      "organization": "EQ-Bench",
      "benchmarkUrl": "https://eqbench.com/creative_writing.html",
      "benchmarkVersion": "Creative Writing v3 · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Elo",
      "scoreType": "elo",
      "unit": "elo",
      "precision": 1,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-benchmarklist-kimi-k3",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 370,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-agents-last-exam",
      "benchmarkId": "agents-last-exam",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "Agents' Last Exam",
      "description": "Agents' Last Exam CLI agent evaluation.",
      "domain": [
        "agents",
        "tool use",
        "reasoning"
      ],
      "organization": "Snorkel AI",
      "benchmarkUrl": "https://snorkel.ai/leaderboard/agents-last-exam/",
      "benchmarkVersion": "Agents' Last Exam · Kimi K3 source snapshot",
      "benchmarkVersionDate": "2026-08-15T00:00:00.000Z",
      "metricName": "Pass rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-snorkel-agents-last-exam",
      "methodology": "A Kimi K3 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 175,
      "coverageNote": "Kimi K3 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Kimi K3 source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-paperbench",
      "benchmarkId": "paperbench",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "PaperBench (Replication Score)",
      "description": "PaperBench research-paper replication evaluation.",
      "domain": [
        "research",
        "professional work"
      ],
      "organization": "PaperBench authors",
      "benchmarkUrl": "https://paperbench.org/",
      "benchmarkVersion": "PaperBench (Replication Score) · Qwen3.8 Max source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Replication score",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "methodology": "A Qwen3.8 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 150,
      "coverageNote": "Qwen3.8 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Qwen3.8 Max source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-qwenreactbench",
      "benchmarkId": "qwenreactbench",
      "auditState": "verified",
      "categoryId": "multimodal",
      "label": "QwenReactBench (Elo)",
      "description": "QwenReactBench multimodal interaction benchmark.",
      "domain": [
        "multimodal",
        "computer use"
      ],
      "organization": "Qwen",
      "benchmarkUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "benchmarkVersion": "QwenReactBench (Elo) · Qwen3.8 Max source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Elo",
      "scoreType": "elo",
      "unit": "elo",
      "precision": 0,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "methodology": "A Qwen3.8 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 160,
      "coverageNote": "Qwen3.8 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Qwen3.8 Max source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-coworkbench",
      "benchmarkId": "coworkbench",
      "auditState": "verified",
      "categoryId": "professional-work",
      "label": "CoWorkBench",
      "description": "Collaborative workplace-agent benchmark.",
      "domain": [
        "professional work",
        "agents"
      ],
      "organization": "Qwen",
      "benchmarkUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "benchmarkVersion": "CoWorkBench · Qwen3.8 Max source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "methodology": "A Qwen3.8 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 170,
      "coverageNote": "Qwen3.8 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Qwen3.8 Max source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-erqa",
      "benchmarkId": "erqa",
      "auditState": "verified",
      "categoryId": "multimodal",
      "label": "ERQA",
      "description": "ERQA multimodal reasoning evaluation.",
      "domain": [
        "multimodal",
        "reasoning"
      ],
      "organization": "Qwen",
      "benchmarkUrl": "https://openlm.ai/qwen3.8/",
      "benchmarkVersion": "ERQA · Qwen3.8 Max source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "methodology": "A Qwen3.8 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 180,
      "coverageNote": "Qwen3.8 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Qwen3.8 Max source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-lvbench",
      "benchmarkId": "lvbench",
      "auditState": "verified",
      "categoryId": "multimodal",
      "label": "LVBench (with Memory)",
      "description": "LVBench video-language evaluation with memory.",
      "domain": [
        "multimodal",
        "video",
        "memory"
      ],
      "organization": "Qwen",
      "benchmarkUrl": "https://openlm.ai/qwen3.8/",
      "benchmarkVersion": "LVBench (with Memory) · Qwen3.8 Max source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Accuracy",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "methodology": "A Qwen3.8 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 190,
      "coverageNote": "Qwen3.8 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Qwen3.8 Max source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-vision2web",
      "benchmarkId": "vision2web",
      "auditState": "verified",
      "categoryId": "multimodal",
      "label": "Vision2Web (Avg. Frontend/Webpage/etc.)",
      "description": "Vision2Web frontend and webpage-generation evaluation.",
      "domain": [
        "multimodal",
        "web generation"
      ],
      "organization": "Qwen",
      "benchmarkUrl": "https://openlm.ai/qwen3.8/",
      "benchmarkVersion": "Vision2Web (Avg. Frontend/Webpage/etc.) · Qwen3.8 Max source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Average score",
      "scoreType": "accuracy",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "methodology": "A Qwen3.8 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 200,
      "coverageNote": "Qwen3.8 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Qwen3.8 Max source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-mobileworld",
      "benchmarkId": "mobileworld",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "MobileWorld",
      "description": "MobileWorld mobile-agent task evaluation.",
      "domain": [
        "mobile agents",
        "computer use"
      ],
      "organization": "MobileWorld authors",
      "benchmarkUrl": "https://openlm.ai/qwen3.8/",
      "benchmarkVersion": "MobileWorld · Qwen3.8 Max source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "methodology": "A Qwen3.8 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 210,
      "coverageNote": "Qwen3.8 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Qwen3.8 Max source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-webarena-verified",
      "benchmarkId": "webarena-verified",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "WebArena-Verified",
      "description": "Verified WebArena browser-agent evaluation.",
      "domain": [
        "web agents",
        "browser use"
      ],
      "organization": "WebArena authors",
      "benchmarkUrl": "https://openlm.ai/qwen3.8/",
      "benchmarkVersion": "WebArena-Verified · Qwen3.8 Max source snapshot",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "methodology": "A Qwen3.8 Max result is retained from the requested source-oriented model record. The source does not establish a common evaluator and harness for the other frontier columns.",
      "order": 220,
      "coverageNote": "Qwen3.8 Max is available for this sheet row; other model cells remain governed by their own source records.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: this Qwen3.8 Max source record does not establish a controlled common-harness comparison."
    },
    {
      "id": "audit-exploitbench-v8",
      "benchmarkId": "exploitbench-v8",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "ExploitBench",
      "description": "Capability-ladder evaluation of autonomous exploitation across 41 recent V8 vulnerabilities.",
      "domain": [
        "cybersecurity",
        "agents",
        "vulnerability exploitation"
      ],
      "organization": "ExploitBench authors",
      "benchmarkUrl": "https://arxiv.org/abs/2605.14153",
      "benchmarkVersion": "ExploitBench V8 · AutoNudge arm · 2026-07-24",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "AutoNudge capability rate",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-anthropic-opus-5-system-card",
      "methodology": "ExploitBench decomposes exploitation into 16 mechanically verified capability flags across 41 V8 environments. The AutoNudge arm allows a keep-trying prompt within the same 300-turn conversation; Cap% averages the fraction of available flags captured across environments.",
      "order": 1450,
      "coverageNote": "The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration.",
      "requiredScaffold": "ExploitBench authors’ harness with Anthropic timeout overlay",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    },
    {
      "id": "audit-automationbench-v1-0-6-public",
      "benchmarkId": "automationbench-v1-0-6-public",
      "auditState": "verified",
      "categoryId": "agents-tool-use",
      "label": "AutomationBench v1.0.6 — public",
      "description": "Zapier’s public 600-task business-workflow automation benchmark.",
      "domain": [
        "automation",
        "agents",
        "professional work"
      ],
      "organization": "Zapier",
      "benchmarkUrl": "https://github.com/zapier/AutomationBench",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "benchmarkVersionDate": "2026-08-16T00:00:00.000Z",
      "metricName": "Task success",
      "scoreType": "pass_rate",
      "unit": "percent",
      "precision": 2,
      "direction": "higher",
      "allowLimitedComparability": true,
      "scoreStatus": "displayed",
      "citationId": "citation-automationbench-public",
      "methodology": "The public v1.0.6 series incorporates the null-type handling fix and is kept separate from Zapier’s private held-out leaderboard.",
      "order": 1098,
      "coverageNote": "Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row.",
      "admissionStatus": "excluded",
      "admissionReason": "Display-only row: fewer than five of seven frontier cells are available; rows with incomplete harness metadata also remain non-scoring."
    }
  ],
  "citations": [
    {
      "id": "citation-anthropic-opus-5-models",
      "title": "Claude Opus 5 model overview and effort controls",
      "publisher": "Anthropic",
      "url": "https://platform.claude.com/docs/en/about-claude/models/overview",
      "sourceType": "vendor",
      "publishedAt": "2026-07-24T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-anthropic-pricing",
      "title": "Claude API pricing",
      "publisher": "Anthropic",
      "url": "https://platform.claude.com/docs/en/about-claude/pricing",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-anthropic-fable-5",
      "title": "Introducing Claude Fable 5 and Claude Mythos 5",
      "publisher": "Anthropic",
      "url": "https://www.anthropic.com/news/claude-fable-5-mythos-5",
      "sourceType": "vendor",
      "publishedAt": "2026-06-09T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-openai-sol",
      "title": "GPT-5.6 Sol model documentation",
      "publisher": "OpenAI",
      "url": "https://developers.openai.com/api/docs/models/gpt-5.6-sol",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-openai-luna",
      "title": "GPT-5.6 Luna model documentation",
      "publisher": "OpenAI",
      "url": "https://developers.openai.com/api/docs/models/gpt-5.6-luna",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-xai-grok",
      "title": "Grok 4.6 model documentation",
      "publisher": "SpaceXAI (xAI)",
      "url": "https://docs.x.ai/developers/models/grok-4.6",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-kimi-k3",
      "title": "Kimi K3 model and release repository",
      "publisher": "Moonshot AI",
      "url": "https://github.com/MoonshotAI/Kimi-K3",
      "sourceType": "vendor",
      "publishedAt": "2026-07-16T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-kimi-pricing",
      "title": "Kimi K3 API pricing",
      "publisher": "Moonshot AI",
      "url": "https://platform.kimi.ai/docs/pricing/chat-k3",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-gemini-3-1-pro",
      "title": "Gemini 3.1 Pro announcement and model card",
      "publisher": "Google DeepMind",
      "url": "https://deepmind.google/models/model-cards/gemini-3-1-pro/",
      "sourceType": "vendor",
      "publishedAt": "2026-02-19T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-gemini-pricing",
      "title": "Gemini API pricing",
      "publisher": "Google",
      "url": "https://ai.google.dev/gemini-api/docs/pricing",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "citation-aa-gdpval",
      "title": "GDPval-AA v2 leaderboard",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live HTML leaderboard snapshot read on 2026-08-15; Elo values are re-estimated as the comparison pool changes."
    },
    {
      "id": "citation-aa-methodology",
      "title": "Artificial Analysis intelligence benchmarking methodology",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/methodology/intelligence-benchmarking",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Version history identifies the v4.1.1 snapshot used by the canonical rows."
    },
    {
      "id": "citation-aa-gpqa",
      "title": "GPQA Diamond leaderboard",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Configuration labels and values are retained per the live leaderboard snapshot; rows without a public value remain missing."
    },
    {
      "id": "citation-prbench-legal",
      "title": "Professional Reasoning Benchmark: Legal",
      "publisher": "Scale Labs",
      "url": "https://labs.scale.com/leaderboard/prbench-legal",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate only: the public board does not cover the full requested roster."
    },
    {
      "id": "citation-gdp-pdf",
      "title": "GDP.pdf professional multimodal reasoning leaderboard",
      "publisher": "Surge AI",
      "url": "https://surgehq.ai/leaderboards/gdp-pdf",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate only: current page lacks a stable public version string and omits Grok 4.6."
    },
    {
      "id": "citation-facts-grounding",
      "title": "The FACTS Leaderboard paper",
      "publisher": "Google Research",
      "url": "https://arxiv.org/abs/2512.10791",
      "sourceType": "research_paper",
      "publishedAt": "2025-12-12T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate only: current public leaderboard does not expose requested-model values."
    },
    {
      "id": "citation-simpleqa",
      "title": "SimpleQA benchmark paper",
      "publisher": "OpenAI",
      "url": "https://arxiv.org/abs/2411.04368",
      "sourceType": "research_paper",
      "publishedAt": "2024-11-06T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate only: the current hosted leaderboard is client-rendered and values are not extractable in the ingestion environment."
    },
    {
      "id": "citation-aa-omniscience",
      "title": "AA-Omniscience knowledge and hallucination benchmark",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/evaluations/omniscience",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate until the structured API snapshot is captured and pinned for all requested configurations."
    },
    {
      "id": "citation-confidencebench",
      "title": "ConfidenceBench calibration leaderboard",
      "publisher": "ConfidenceBench",
      "url": "https://confidencebench.com/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate only: one requested model is present and the evaluation set is private."
    },
    {
      "id": "citation-longfact",
      "title": "Long-form factuality evaluation with LongFact and SAFE",
      "publisher": "Google Research",
      "url": "https://arxiv.org/abs/2403.18802",
      "sourceType": "research_paper",
      "publishedAt": "2024-03-28T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate only: a real metric exists but requested-model public result coverage is insufficient."
    },
    {
      "id": "citation-tau3-banking",
      "title": "τ³-Banking benchmark leaderboard",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/evaluations/tau3-banking",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate only: the live snapshot changes and the requested Opus effort ladder is incomplete."
    },
    {
      "id": "citation-aa-performance",
      "title": "Artificial Analysis language model API performance methodology",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/methodology/performance-benchmarking",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate only: rolling P50 performance metrics are separate from capability scoring."
    },
    {
      "id": "citation-legalbench",
      "title": "LegalBench: a collaboratively built benchmark for measuring legal reasoning",
      "publisher": "LegalBench",
      "url": "https://arxiv.org/abs/2308.11432",
      "sourceType": "research_paper",
      "publishedAt": "2023-08-21T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Used only to document why the legacy generic Legal Workbench label was rejected."
    },
    {
      "id": "citation-arc-prize",
      "title": "ARC-AGI-2 benchmark results",
      "publisher": "ARC Prize Foundation",
      "url": "https://arcprize.org/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Existing fixture row audited; the current requested roster does not have a complete structured source snapshot."
    },
    {
      "id": "citation-livecodebench",
      "title": "LiveCodeBench benchmark",
      "publisher": "LiveCodeBench",
      "url": "https://livecodebench.github.io/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Existing fixture row audited; requested-model coverage is not sufficiently extractable."
    },
    {
      "id": "citation-swebench",
      "title": "SWE-bench Verified benchmark",
      "publisher": "SWE-bench",
      "url": "https://www.swebench.com/verified.html",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Existing fixture row audited; current public values are harness-sensitive and not complete across the requested roster."
    },
    {
      "id": "citation-frontiermath",
      "title": "FrontierMath benchmark",
      "publisher": "Epoch AI",
      "url": "https://epoch.ai/frontiermath",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Existing fixture row audited; current requested-model values are not sufficiently extractable and version-pinned."
    },
    {
      "id": "citation-aime",
      "title": "AIME benchmark archive",
      "publisher": "American Invitational Mathematics Examination",
      "url": "https://artofproblemsolving.com/wiki/index.php/AIME",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Existing fixture row audited; no current requested-model result source was verified."
    },
    {
      "id": "citation-cursorbench-32",
      "title": "CursorBench 3.2 leaderboard",
      "publisher": "Cursor",
      "url": "https://cursor.com/cursorbench",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "First-party leaderboard snapshot; primary score and secondary cost, token, and step metrics are retained separately."
    },
    {
      "id": "citation-deepswe-11",
      "title": "DeepSWE v1.1 leaderboard and revision notes",
      "publisher": "Datacurve",
      "url": "https://deepswe.datacurve.ai/artifacts/v1.1/leaderboard-live.json",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "contentHash": "sha256-a956de6fc305b056d57fd464fd3584fdb58498ade5aaec23ca77760755a0c291",
      "accessNote": "Hash covers the captured raw JSON artifact used for this dataset revision."
    },
    {
      "id": "citation-frontiercode-11",
      "title": "FrontierCode benchmark",
      "publisher": "Cognition",
      "url": "https://cognition.ai/blog/frontier-code",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Extended, Main, and Diamond subsets are kept as distinct source metrics; source scores are reported per effort setting."
    },
    {
      "id": "citation-apex-agents",
      "title": "APEX-Agents leaderboard",
      "publisher": "Mercor",
      "url": "https://www.mercor.com/apex/apex-agents-leaderboard/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Leaderboard distinguishes Pass@1 from mean rubric score; the primary matrix metric is Pass@1."
    },
    {
      "id": "citation-apex-swe",
      "title": "APEX-SWE leaderboard",
      "publisher": "Mercor and Cognition",
      "url": "https://www.mercor.com/apex/apex-swe-leaderboard/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Professional software-engineering benchmark with 200 held-out cases and a published Pass@1 leaderboard."
    },
    {
      "id": "citation-aa-briefcase",
      "title": "AA-Briefcase leaderboard and methodology",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/articles/aa-briefcase",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Overall AA-Briefcase Elo is a composite of rubric, analytical-quality, and presentation-quality signals; those layers are not treated as separate benchmark rows."
    },
    {
      "id": "citation-harvey-lab-aa",
      "title": "Harvey LAB-AA leaderboard",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/evaluations/harvey-lab-aa",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "The page reports both criterion pass rate and all-pass rate; this dataset uses criterion pass rate as the primary comparable metric."
    },
    {
      "id": "citation-aa-briefcase-secondary",
      "title": "AA-Briefcase launch result summary",
      "publisher": "eesel AI",
      "url": "https://www.eesel.ai/blog/aa-briefcase",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Used only as a dated secondary result source for the launch-era Fable 5 Elo; the benchmark definition remains the Artificial Analysis canonical page."
    },
    {
      "id": "citation-terminal-bench",
      "title": "Terminal-Bench continuous benchmark and leaderboard",
      "publisher": "Terminal-Bench / Harbor",
      "url": "https://www.tbench.ai/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Candidate source for Terminal-Bench v2.1 and v3.0; versions are kept separate and neither is mixed into the admitted TLS rows."
    },
    {
      "id": "citation-physunibench",
      "title": "PhysUniBench benchmark and project page",
      "publisher": "PrismaX Team",
      "url": "https://prismax-team.github.io/PhysUniBenchmark/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Canonical project page documents 3,304 multimodal physics questions, eight sub-disciplines, and separate multiple-choice/open-ended accuracy results."
    },
    {
      "id": "citation-humanitys-last-exam",
      "title": "Humanity's Last Exam",
      "publisher": "Center for AI Safety and collaborators",
      "url": "https://lastexam.ai/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Canonical benchmark page documents the expert-vetted question set, accuracy, and calibration-error metrics."
    },
    {
      "id": "citation-aime-2026",
      "title": "AIME problem archive and 2026 contest sets",
      "publisher": "American Invitational Mathematics Examination",
      "url": "https://artofproblemsolving.com/wiki/index.php/AIME",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "The problem archive is the canonical public source for the contest-year problem sets; model results remain candidate until prompt and tool settings are pinned."
    },
    {
      "id": "citation-chembench",
      "title": "ChemBench chemistry benchmark leaderboard",
      "publisher": "LamaLab",
      "url": "https://chembench.lamalab.org/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Public leaderboard reports overall and topic-level accuracy across the ChemBench suite."
    },
    {
      "id": "citation-lab-bench",
      "title": "LAB-Bench biology research benchmark",
      "publisher": "FutureHouse",
      "url": "https://github.com/Future-House/LAB-Bench",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Project repository documents the eight categories, 30 subtasks, public subset, and private contamination-monitoring subset."
    },
    {
      "id": "citation-discoverphysics",
      "title": "DiscoverPhysics leaderboard",
      "publisher": "DiscoverPhysics authors",
      "url": "https://sampsonml.github.io/DiscoverPhysicsLeaderboard/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Official leaderboard defines Pass@k, normalized trajectory MSE, and explanation score for simulated-world discovery."
    },
    {
      "id": "citation-sci-code",
      "title": "SciCode benchmark leaderboard",
      "publisher": "SciCode authors",
      "url": "https://scicode-bench.github.io/leaderboard/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Maintainer leaderboard reports main-problem resolve rate and subproblem performance separately."
    },
    {
      "id": "citation-terminal-bench-2-0",
      "title": "Terminal-Bench 2.0 leaderboard",
      "publisher": "Terminal-Bench / Harbor",
      "url": "https://www.tbench.ai/leaderboard/terminal-bench/2.0",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Version-specific official leaderboard; v2.0 is kept separate from the corrected v2.1 release."
    },
    {
      "id": "citation-charxiv",
      "title": "CharXiv chart understanding benchmark",
      "publisher": "Princeton NLP",
      "url": "https://charxiv.github.io/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Canonical project page documents the v1 dataset, descriptive/reasoning tracks, and public leaderboard."
    },
    {
      "id": "citation-tau2-bench",
      "title": "τ-bench and τ²-bench leaderboard",
      "publisher": "Sierra Research",
      "url": "https://taubench.com/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live leaderboard separates τ² text domains from later τ³ knowledge and voice variants."
    },
    {
      "id": "citation-bfcl-v4",
      "title": "Berkeley Function Calling Leaderboard v4",
      "publisher": "Berkeley Gorilla team",
      "url": "https://gorilla.cs.berkeley.edu/leaderboard.html",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "BFCL v4's overall score and five weighted segments are documented by the benchmark maintainer."
    },
    {
      "id": "citation-browsecomp",
      "title": "BrowseComp: a benchmark for browsing agents",
      "publisher": "OpenAI",
      "url": "https://openai.com/index/browsecomp",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "OpenAI's benchmark release documents the 1,266-question set and exact-answer accuracy; public results are scaffold and search-budget dependent."
    },
    {
      "id": "citation-osworld-2",
      "title": "OSWorld 2.0 benchmark",
      "publisher": "xlang AI and collaborators",
      "url": "https://osworld-v2.xlang.ai/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Official page documents the 108 workflows, 500-step budget, binary completion, and partial checkpoint metrics."
    },
    {
      "id": "citation-online-mind2web",
      "title": "Online Mind2Web leaderboard",
      "publisher": "Ohio State University NLP Group",
      "url": "https://huggingface.co/spaces/osunlp/Online_Mind2Web_Leaderboard",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Official leaderboard describes 300 tasks across 136 live websites and separates automatic WebJudge from human evaluation."
    },
    {
      "id": "citation-androidworld",
      "title": "AndroidWorld benchmark",
      "publisher": "Google Research",
      "url": "https://google-research.github.io/android_world/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Project page documents 116 dynamically instantiated tasks across 20 Android apps and durable state-based reward signals."
    },
    {
      "id": "citation-faith-eval",
      "title": "FaithEval contextual faithfulness benchmark",
      "publisher": "Salesforce AI Research",
      "url": "https://github.com/SalesforceAIResearch/FaithEval",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Repository documents unanswerable, inconsistent, and counterfactual context tasks and their separate accuracy summaries."
    },
    {
      "id": "citation-healthbench",
      "title": "HealthBench: Evaluating Large Language Models",
      "publisher": "OpenAI",
      "url": "https://cdn.openai.com/pdf/bd7a39d5-9e9f-47b3-903c-8b847ca650c7/healthbench_paper.pdf",
      "sourceType": "research_paper",
      "publishedAt": "2025-05-12T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "The paper defines HealthBench, Consensus, Hard, physician rubrics, and mean normalized scoring."
    },
    {
      "id": "citation-plawbench",
      "title": "PLawBench: A Rubric-Based Benchmark for Evaluating LLMs in Real-World Legal Practice",
      "publisher": "Association for Computational Linguistics",
      "url": "https://aclanthology.org/2026.acl-long.458/",
      "sourceType": "research_paper",
      "publishedAt": "2026-07-01T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "ACL paper defines the 850-question benchmark, three legal-practice task groups, rubric items, and scoring-rate construction."
    },
    {
      "id": "citation-ifhierbench",
      "title": "IFHierBench: Hierarchical Instruction Following",
      "publisher": "IFHierBench authors",
      "url": "https://arxiv.org/html/2607.27912v1",
      "sourceType": "research_paper",
      "publishedAt": "2026-07-31T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Paper defines 600 prompts, four constraint-tree depths, 35 constraints, and deterministic hierarchical checkers."
    },
    {
      "id": "citation-longbench-v2",
      "title": "LongBench v2",
      "publisher": "LongBench authors",
      "url": "https://longbench2.github.io/",
      "sourceType": "research_paper",
      "publishedAt": "2025-01-01T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Official leaderboard defines zero-shot and CoT settings, compensated accuracy, and context-length bins."
    },
    {
      "id": "citation-helmet",
      "title": "HELMET long-context benchmark",
      "publisher": "Princeton NLP",
      "url": "https://princeton-nlp.github.io/HELMET/",
      "sourceType": "research_paper",
      "publishedAt": "2024-10-03T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Project page defines seven application-centric task categories and controllable context lengths."
    },
    {
      "id": "citation-ruler",
      "title": "RULER long-context benchmark",
      "publisher": "NVIDIA",
      "url": "https://github.com/NVIDIA/RULER",
      "sourceType": "research_paper",
      "publishedAt": "2024-04-08T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Official repository defines 13 generated tasks across retrieval, multi-hop tracing, aggregation, and question answering."
    },
    {
      "id": "citation-memory-agent-bench",
      "title": "MemoryAgentBench",
      "publisher": "HUST AI and collaborators",
      "url": "https://github.com/HUST-AI-HYZ/MemoryAgentBench",
      "sourceType": "research_paper",
      "publishedAt": "2025-07-07T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Repository and paper define four memory competencies and the task-specific accuracy/recall metrics."
    },
    {
      "id": "citation-strongreject",
      "title": "A StrongREJECT for Empty Jailbreaks",
      "publisher": "StrongREJECT authors",
      "url": "https://arxiv.org/abs/2402.10260",
      "sourceType": "research_paper",
      "publishedAt": "2024-02-15T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Paper defines refusal, specificity, convincingness, and the composite StrongREJECT score."
    },
    {
      "id": "citation-agentharm",
      "title": "AgentHarm: A Benchmark for Measuring Harmfulness of LLM Agents",
      "publisher": "AI Safety Institute and collaborators",
      "url": "https://arxiv.org/abs/2410.09024",
      "sourceType": "research_paper",
      "publishedAt": "2024-10-11T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Paper and public dataset define HarmScore, RefusalRate, NonRefusalHarmScore, and the released behavior subsets."
    },
    {
      "id": "citation-agentdojo",
      "title": "AgentDojo prompt-injection benchmark",
      "publisher": "ETH Zurich SPY Lab",
      "url": "https://agentdojo.spylab.ai/",
      "sourceType": "research_paper",
      "publishedAt": "2024-06-19T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Project documentation defines benign utility, utility under attack, and security success metrics over dynamic tool-use suites."
    },
    {
      "id": "citation-cybench",
      "title": "Cybench cybersecurity agent benchmark",
      "publisher": "Cybench authors",
      "url": "https://cybench.github.io/",
      "sourceType": "research_paper",
      "publishedAt": "2024-08-13T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Official site defines unguided, subtask-guided, and subtask performance across 40 cybersecurity tasks."
    },
    {
      "id": "citation-mask",
      "title": "The MASK Benchmark",
      "publisher": "Center for AI Safety and collaborators",
      "url": "https://www.mask-benchmark.ai/",
      "sourceType": "research_paper",
      "publishedAt": "2025-03-05T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Official project page defines belief elicitation, pressure prompts, and honesty versus accuracy."
    },
    {
      "id": "citation-marco-bench-mif",
      "title": "Marco-Bench-MIF",
      "publisher": "Association for Computational Linguistics",
      "url": "https://aclanthology.org/2025.acl-long.1172/",
      "sourceType": "research_paper",
      "publishedAt": "2025-07-01T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "ACL paper defines localized instruction following across 30 languages and the language-level accuracy aggregation."
    },
    {
      "id": "citation-global-mmlu",
      "title": "Global-MMLU dataset and benchmark",
      "publisher": "Cohere Labs and collaborators",
      "url": "https://huggingface.co/datasets/CohereLabs/Global-MMLU",
      "sourceType": "research_paper",
      "publishedAt": "2025-01-01T00:00:00.000Z",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Dataset card and paper distinguish full, annotated, culturally sensitive, culturally agnostic, and Lite subsets across 42 languages."
    },
    {
      "id": "citation-eq-bench-4",
      "title": "EQ-Bench 4 leaderboard and methodology",
      "publisher": "EQ-Bench authors",
      "url": "https://eqbench.com/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Official leaderboard defines blind pairwise ability judgments, soft Bradley-Terry Elo, anchors, and bootstrap intervals."
    },
    {
      "id": "citation-chatbot-arena",
      "title": "LMSYS Chatbot Arena policy and evaluation method",
      "publisher": "LMSYS and UC Berkeley collaborators",
      "url": "https://www.lmsys.org/blog/2024-03-01-policy/",
      "sourceType": "independent",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Methodology source for the human pairwise-preference evaluation; live ratings are treated as a dated external check rather than a static benchmark result."
    },
    {
      "id": "citation-vals-code-migration",
      "title": "Code Migration leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/code-migration",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-excel-modeling",
      "title": "Excel Modeling Benchmark leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/emb",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-legal-research",
      "title": "Legal Research Bench leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/legal_research",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-finance-agent-v2",
      "title": "Finance Agent v2 leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/fabv2",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-harvey-lab",
      "title": "Harvey Legal Agent Benchmark · Vals run leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/hlab",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-vibe-code",
      "title": "Vibe Code Bench v1.1 leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/vibe-code",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-swebench",
      "title": "SWE-bench Verified · Vals run leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/swebench",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-mmlu-pro",
      "title": "MMLU-Pro · Vals run leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/mmlu_pro",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-mmmu-pro",
      "title": "MMMU-Pro · Vals run leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/mmmu",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-livecodebench",
      "title": "LiveCodeBench · Vals run leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/lcb",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-vals-programbench",
      "title": "ProgramBench · Vals run leaderboard",
      "publisher": "Vals AI",
      "url": "https://vals.ai/benchmarks/programbench",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Live Vals AI page snapshot captured for dataset ingestion. Model rows, harness labels, and missing cells are preserved as observed; no value is inferred from another source."
    },
    {
      "id": "citation-apex-swe-integration",
      "title": "APEX-SWE · Integration leaderboard",
      "publisher": "Mercor",
      "url": "https://www.mercor.com/apex/apex-swe-leaderboard/integration-swe/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Track-specific public leaderboard snapshot. Exact effort labels are retained; Sol XHigh and Grok High are recorded as missing for the requested Sol Max and Grok XHigh cells."
    },
    {
      "id": "citation-apex-swe-observability",
      "title": "APEX-SWE · Observability leaderboard",
      "publisher": "Mercor",
      "url": "https://www.mercor.com/apex/apex-swe-leaderboard/observability-swe/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-15T00:00:00.000Z",
      "accessNote": "Track-specific public leaderboard snapshot. Exact effort labels are retained; Sol XHigh and Grok High are recorded as missing for the requested Sol Max and Grok XHigh cells."
    },
    {
      "id": "citation-llm-boss-swebench-verified",
      "title": "SWE-bench Verified model scores",
      "publisher": "LLM Boss",
      "url": "https://llm-boss.com/benchmarks/swe-bench-verified",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Comparison page reports Fable 5 and GPT-5.6 Sol values and links the Vals AI leaderboard. Configuration and harness caveats are retained."
    },
    {
      "id": "citation-datacamp-opus-5",
      "title": "Claude Opus 5 benchmark comparison",
      "publisher": "DataCamp",
      "url": "https://www.datacamp.com/blog/claude-opus-5",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Secondary comparison of vendor-reported SWE-bench Verified results. It is not treated as a common-harness evaluation."
    },
    {
      "id": "citation-matharena",
      "title": "MathArena competition leaderboard",
      "publisher": "MathArena",
      "url": "https://matharena.ai/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Live evaluator leaderboard. Month-specific competition rows and model configurations are kept distinct."
    },
    {
      "id": "citation-matharena-fable-5-max",
      "title": "Claude Fable 5 max · MathArena results",
      "publisher": "MathArena",
      "url": "https://matharena.ai/models/anthropic_fable_5_max",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "citation-matharena-sol-max",
      "title": "GPT-5.6 Sol max · MathArena results",
      "publisher": "MathArena",
      "url": "https://matharena.ai/models/openai_gpt_56_sol",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "citation-matharena-opus-5-max",
      "title": "Claude Opus 5 max · MathArena results",
      "publisher": "MathArena",
      "url": "https://matharena.ai/models/anthropic_opus_5_max",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "citation-together-frontier-comparison",
      "title": "Frontier model performance benchmark comparison",
      "publisher": "Together AI",
      "url": "https://www.together.ai/models/glm-52",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "The same comparison table reports SciCode, FrontierCode, and Terminal-Bench 2.1 values. The page does not fully identify every underlying harness or reasoning setting."
    },
    {
      "id": "citation-benchmarklist-fable-5",
      "title": "Claude Fable 5 benchmark scores and source provenance",
      "publisher": "BenchmarkList",
      "url": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "sourceType": "independent",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Source-oriented benchmark index used for the requested three-model Vals, arena, and cross-source comparison rows. Per-row evaluator and harness labels are retained."
    },
    {
      "id": "citation-tosea-opus-5-head-to-head",
      "title": "Claude Opus 5 benchmark head-to-head",
      "publisher": "Tosea.ai",
      "url": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Transcribes Anthropic’s published Opus 5 comparison table. The five head-to-head rows are kept distinct from independent leaderboards."
    },
    {
      "id": "citation-vals-corpfin-v2",
      "title": "CorpFin v2 benchmark",
      "publisher": "Vals AI",
      "url": "https://www.vals.ai/benchmarks/corp_fin_v2",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Official Vals benchmark definition and archived leaderboard; results use a shared Vals evaluation family."
    },
    {
      "id": "citation-benchmarklist-kimi-k3",
      "title": "Kimi K3 benchmark scores and source provenance",
      "publisher": "BenchmarkList",
      "url": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "sourceType": "independent",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Source-oriented Kimi K3 model record used for the sheet additions; individual benchmark versions and provisional caveats are retained."
    },
    {
      "id": "citation-matharena-kimi-k3",
      "title": "Kimi K3 · MathArena results",
      "publisher": "MathArena",
      "url": "https://matharena.ai/models/moonshot_k3",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "citation-explainx-terminal-bench-3",
      "title": "GLM-5.3 benchmark comparison",
      "publisher": "ExplainX",
      "url": "https://explainx.ai/blog/glm-5-3-launch-cyber-defense-benchmarks-august-2026",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Third-party transcription of the Z.ai comparison snapshot. Terminal-Bench 3.0 is retained with an explicit snapshot mismatch caveat."
    },
    {
      "id": "citation-snorkel-agents-last-exam",
      "title": "Agents' Last Exam leaderboard",
      "publisher": "Snorkel AI",
      "url": "https://snorkel.ai/leaderboard/agents-last-exam/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Official leaderboard rows matching the table pass-rate configuration: GPT-5.6 Sol at 30.6%, Kimi K3 at 28.3%, and Qwen3.8 Max at 27.0%. The separate Qwen score metric is not substituted."
    },
    {
      "id": "citation-grok-4-6-benchmark-batch",
      "title": "Grok 4.6 High benchmark comparison batch",
      "publisher": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "url": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "sourceType": "independent",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Source-matched Grok 4.6 High additions from the supplied benchmark pages. Values are kept at the requested historical snapshot and derived values remain explicitly marked."
    },
    {
      "id": "citation-grok-4-6-arena",
      "title": "Grok 4.6 High LMArena text score",
      "publisher": "Arena AI",
      "url": "https://arena.ai/leaderboard/text",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Preliminary/live LMArena text score; it is expected to move as additional votes accumulate."
    },
    {
      "id": "citation-qwen-3-8-benchmark-batch",
      "title": "Qwen3.8 Max benchmark comparison batch",
      "publisher": "BenchmarkList, OpenLM.ai, and Artificial Analysis",
      "url": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "sourceType": "independent",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Source-matched Qwen3.8 Max additions from the supplied model-card, benchmark-index, and multimodal comparison pages. Distinct benchmark configurations and metric caveats are retained."
    },
    {
      "id": "citation-qwen-3-8-vals",
      "title": "Qwen3.8 Max Vals benchmark snapshot",
      "publisher": "Vals AI / BenchLM",
      "url": "https://benchlm.ai/benchmarks/valsswebench",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Vals-hosted benchmark snapshots used for the Qwen3.8 Max comparison rows; the requested historical peer snapshot is kept separate from newer live composite values."
    },
    {
      "id": "citation-qwen-3-8-osworld",
      "title": "Qwen3.8 Max OSWorld 2.0 partial score",
      "publisher": "Forward Future",
      "url": "https://signals.forwardfuture.com/qwen3-8-benchmarks/",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Qwen reports OSWorld 2.0 as 19.4 binary / 46.7 partial; the partial metric is used to match the existing Fable and GPT comparison row."
    },
    {
      "id": "citation-zai-glm-5-3",
      "title": "GLM-5.3 launch benchmark results",
      "publisher": "Z.ai",
      "url": "https://z.ai/blog/glm-5.3",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Provider-published GLM-5.3 results. Agents’ Last Exam is retained as a CLI pass rate, while Terminal-Bench 2.1 is marked as a provider-specific Claude Code run."
    },
    {
      "id": "citation-deepseek-v4-pro-0813",
      "title": "DeepSeek V4 Pro 0813 benchmark results",
      "publisher": "DeepSeek",
      "url": "https://api-docs.deepseek.com/news/news260813/",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "DeepSeek V4 Pro 0813 provider results. Vendor and independent evaluator rows remain separate, and multimodal or incompatible harness values are not substituted."
    },
    {
      "id": "citation-deepseek-aa",
      "title": "DeepSeek V4 Pro 0813 Artificial Analysis results",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1",
      "sourceType": "independent",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Artificial Analysis scores used where the table row is explicitly an AA evaluation, including Terminal-Bench 2.1."
    },
    {
      "id": "citation-deepseek-vals",
      "title": "DeepSeek V4 Pro 0813 Vals benchmark results",
      "publisher": "Vals AI / BenchLM",
      "url": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Vals-hosted scores used for rows whose existing peer values come from the Vals evaluation family."
    },
    {
      "id": "citation-deepseek-arena",
      "title": "DeepSeek V4 Pro 0813 LMArena text score",
      "publisher": "Arena AI",
      "url": "https://arena.ai/leaderboard/text",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "AutoEval text Elo for the deepseek-v4-pro-max-20260813 entry; the reported uncertainty is approximately ±10 Elo."
    },
    {
      "id": "citation-muse-spark-1-2-aa",
      "title": "Muse Spark 1.2 Artificial Analysis evaluations",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "sourceType": "independent",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Current Artificial Analysis Muse Spark 1.2 results. Launch-day values and later leaderboard drift are not conflated."
    },
    {
      "id": "citation-muse-spark-1-2-vals",
      "title": "Muse Spark 1.2 Vals benchmark results",
      "publisher": "Vals AI / BenchLM",
      "url": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Vals scores use the evaluator family and snapshot matching the existing Vals rows. The older Vals Index version is explicitly marked with a dagger."
    },
    {
      "id": "citation-muse-spark-1-2-code",
      "title": "Muse Spark 1.2 Muse Code benchmark results",
      "publisher": "BenchLM",
      "url": "https://benchlm.ai/compare/muse-spark-1-1-vs-muse-spark-1-2",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "DeepSWE and Terminal-Bench 2.1 are Muse Spark 1.2 running with Muse Code; these are model-plus-agent results and retain an asterisk."
    },
    {
      "id": "citation-muse-spark-1-2-arena",
      "title": "Muse Spark 1.2 LMArena text score",
      "publisher": "BenchmarkList",
      "url": "https://benchmarklist.com/models/meta-muse-spark-1.2/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Muse Spark 1.2 text Elo is rounded from the reported 1498.64 score."
    },
    {
      "id": "citation-muse-spark-1-2-livebench",
      "title": "Muse Spark 1.2 LiveBench result",
      "publisher": "LiveBench",
      "url": "https://livebench.ai/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Muse Spark 1.2 LiveBench score at extra-high reasoning effort."
    },
    {
      "id": "citation-gemini-3-7-flash-google",
      "title": "Gemini 3.7 Flash benchmark results",
      "publisher": "Google DeepMind",
      "url": "https://deepmind.google/models/gemini/flash/",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Google-reported Gemini 3.7 Flash high-reasoning results. Protocol-specific tool and agent configurations remain explicitly annotated."
    },
    {
      "id": "citation-gemini-3-7-flash-aa",
      "title": "Gemini 3.7 Flash Artificial Analysis results",
      "publisher": "Artificial Analysis",
      "url": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "sourceType": "independent",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Current high-reasoning Artificial Analysis results, including Intelligence Index, HLE, GPQA, AA-Omniscience, AA-LCR, and SciCode."
    },
    {
      "id": "citation-gemini-3-7-flash-vals",
      "title": "Gemini 3.7 Flash Vals benchmark results",
      "publisher": "Vals AI / BenchLM",
      "url": "https://www.vals.ai/models/google_gemini-3.7-flash",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Vals scores use the named benchmark-family result. The Vals Index entry is marked with an asterisk because 59.31% belongs to Vals Index v2 rather than the older peer snapshot."
    },
    {
      "id": "citation-gemini-3-7-flash-arena",
      "title": "Gemini 3.7 Flash LMArena text score",
      "publisher": "Arena AI",
      "url": "https://lmarena.ai/leaderboard/text",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Preliminary Gemini 3.7 Flash text Arena Elo with approximately ±8 Elo uncertainty."
    },
    {
      "id": "citation-gemini-3-7-flash-livebench",
      "title": "Gemini 3.7 Flash LiveBench result",
      "publisher": "LiveBench",
      "url": "https://livebench.ai/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Current Gemini 3.7 Flash LiveBench overall score."
    },
    {
      "id": "citation-qwen-3-8-vals-index",
      "title": "Qwen3.8 Max early Vals Index listing",
      "publisher": "Memeburn",
      "url": "https://memeburn.com/kimi-k3-vs-qwen-3-8/",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Reports the early Vals Index listing at 66.12%; Vals AI’s public launch announcement independently corroborates the value rounded to 66.1%. This historical result is kept separate from Vals Index v2."
    },
    {
      "id": "citation-qwen-3-8-livebench",
      "title": "Qwen3.8 Max LiveBench result",
      "publisher": "LiveBench",
      "url": "https://livebench.ai/",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "LiveBench-2026-06-25 live overall leaderboard result for Qwen3.8 Max. This live value may move when the benchmark refreshes."
    },
    {
      "id": "citation-muse-spark-1-2-toolathlon",
      "title": "Muse Spark 1.2 Toolathlon-Verified result",
      "publisher": "Toolathlon",
      "url": "https://toolathlon.xyz/docs/leaderboard",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Independently evaluated Toolathlon-Verified Pass@1 result for Muse Spark 1.2 at extra-high reasoning effort with the default agent, dated 2026-08-05."
    },
    {
      "id": "citation-anthropic-opus-5-system-card",
      "title": "Claude Opus 5 System Card",
      "publisher": "Anthropic",
      "url": "https://www-cdn.anthropic.com/b514064af1408018e64b1ad24e7d5e75850b4ffd/Claude%20Opus%205%20System%20Card.pdf",
      "sourceType": "vendor",
      "publishedAt": "2026-07-24T00:00:00.000Z",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Reports ExploitBench AutoNudge Cap% at 70 for Claude Opus 5. The standard Opus 5 evaluation configuration is adaptive thinking at max effort unless otherwise noted."
    },
    {
      "id": "citation-deepseek-v4-pro-card",
      "title": "DeepSeek V4 Pro model card",
      "publisher": "DeepSeek",
      "url": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro",
      "sourceType": "vendor",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "The model card reports MCPAtlas Public Pass@1 at 73.6 for DeepSeek V4 Pro Max and identifies the Max reasoning configuration explicitly."
    },
    {
      "id": "citation-kimi-k3-frontiercode-main",
      "title": "Kimi K3 FrontierCode result",
      "publisher": "Together AI",
      "url": "https://www.together.ai/models/kimi-k3",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Reports Kimi K3 at 44.2% on FrontierCode in a table whose peer values match the FrontierCode v1.1 Main public leaderboard; the page states Kimi K3 runs at maximum thinking effort."
    },
    {
      "id": "citation-grok-4-6-frontiercode-main",
      "title": "Grok 4.6 FrontierCode v1.1 Main result",
      "publisher": "Digital Applied",
      "url": "https://www.digitalapplied.com/blog/vendor-benchmark-tables-reading-disclosed-losses-2026",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Transcribes the Cognition Main leaderboard result for Grok 4.6 High at 48.0%, kept separate from xAI’s 61.3% Extended-subset result."
    },
    {
      "id": "citation-deepseek-frontiercode-main",
      "title": "DeepSeek V4 Pro Max FrontierCode v1.1 result",
      "publisher": "LLM Stats",
      "url": "https://llm-stats.com/benchmarks/frontiercode-1.1",
      "sourceType": "third_party",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Reports DeepSeek V4 Pro Max at 0.176 on the FrontierCode v1.1 leaderboard. This secondary mirror labels the result unverified, so the observation remains display-only."
    },
    {
      "id": "citation-gemini-3-7-frontiercode-main",
      "title": "Gemini 3.7 Flash model card",
      "publisher": "Google DeepMind",
      "url": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "sourceType": "vendor",
      "publishedAt": "2026-08-13T00:00:00.000Z",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Reports Gemini 3.7 Flash at 43.6% on FrontierCode 1.1 Main under Google’s published evaluation configuration."
    },
    {
      "id": "citation-frontiercode-11-data",
      "title": "FrontierCode v1.1 leaderboard data",
      "publisher": "Cognition",
      "url": "https://cognition.com/data/frontiercode-leaderboard/data.json",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Official leaderboard payload. The Extended subset reports weighted score separately from blocker-clearing pass rate and records per-row effort labels."
    },
    {
      "id": "citation-automationbench-public",
      "title": "AutomationBench public v1.0.6 leaderboard",
      "publisher": "Zapier",
      "url": "https://github.com/zapier/AutomationBench",
      "sourceType": "benchmark_org",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Official public 600-task benchmark. The Opus 5 max score is 50.3%; the separate 26.0% result belongs to the private held-out evaluation."
    },
    {
      "id": "citation-ouroboros-osworld-opus-5",
      "title": "Ouroboros on OSWorld-Verified with Claude Opus 5",
      "publisher": "Ouroboros",
      "url": "https://huggingface.co/datasets/razzant/ouroboros-osworld-verified-opus5",
      "sourceType": "third_party",
      "publishedAt": "2026-08-11T00:00:00.000Z",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Self-reported but fully documented 361-task run: Opus 5 at max acting effort, screenshot-only, one rollout, 100 policy turns, and the official evaluator."
    },
    {
      "id": "citation-meta-muse-spark-1-2-methodology",
      "title": "Muse Spark 1.2 and Muse Code evaluation methodology",
      "publisher": "Meta",
      "url": "https://research.meta.ai/static/muse-spark-1-2-methodology",
      "sourceType": "vendor",
      "publishedAt": "2026-08-05T00:00:00.000Z",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Reports Muse Spark 1.2 at 90.3% on MCP Atlas using xhigh effort, Meta Model API, and the benchmark provider’s own harness and scoring pipeline."
    },
    {
      "id": "citation-grok-4-6-model-card",
      "title": "Grok 4.6 model card",
      "publisher": "SpaceXAI (xAI)",
      "url": "https://media.x.ai/v1/website/card-7f81d41b.pdf",
      "sourceType": "vendor",
      "publishedAt": "2026-08-12T00:00:00.000Z",
      "retrievedAt": "2026-08-16T00:00:00.000Z",
      "accessNote": "Reports Grok 4.6 High at 63.2% accuracy on OfficeQA Pro."
    }
  ],
  "observations": [
    {
      "id": "observation-gdpval-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1849,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1827,
        "ciUpper": 1871,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-opus-5-xhigh",
      "fixture": false,
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1817,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1796,
        "ciUpper": 1838,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-opus-5-high",
      "fixture": false,
      "modelId": "claude-opus-5-high",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1735,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "high effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1715,
        "ciUpper": 1755,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-opus-5-medium",
      "fixture": false,
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1621,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "medium effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1602,
        "ciUpper": 1640,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-opus-5-low",
      "fixture": false,
      "modelId": "claude-opus-5-low",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1456,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "low effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1437,
        "ciUpper": 1475,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-fable-5",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1739,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1723,
        "ciUpper": 1755,
        "confidenceLevel": 0.9
      },
      "comparability": "limited",
      "comparabilityNote": "Artificial Analysis lists a fallback behavior for a subset of safety-classified prompts; retained as limited comparability.",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-sol",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1725,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1709,
        "ciUpper": 1741,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-luna",
      "fixture": false,
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1581,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "gpt-5.6-luna",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1565,
        "ciUpper": 1597,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-grok",
      "fixture": false,
      "modelId": "grok-4-6-high",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1746,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 1726,
        "ciUpper": 1766,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gdpval-gemini",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 965,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gdpval",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GDPval-AA v2 leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gdpval-aa",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "thinking",
      "reasoningBudget": "high",
      "scaffold": "Artificial Analysis agent harness",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 949,
        "ciUpper": 981,
        "confidenceLevel": 0.9
      },
      "comparability": "comparable",
      "notes": "Native Elo rating retained; no conversion to percent."
    },
    {
      "id": "observation-gpqa-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 93.2,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "No tools",
      "comparability": "comparable"
    },
    {
      "id": "observation-gpqa-opus-5-xhigh",
      "fixture": false,
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 93.7,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "xhigh effort",
      "scaffold": "No tools",
      "comparability": "comparable"
    },
    {
      "id": "observation-gpqa-opus-5-high",
      "fixture": false,
      "modelId": "claude-opus-5-high",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 93.7,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "high effort",
      "scaffold": "No tools",
      "comparability": "comparable"
    },
    {
      "id": "observation-gpqa-opus-5-medium",
      "fixture": false,
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 91.9,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "medium effort",
      "scaffold": "No tools",
      "comparability": "comparable"
    },
    {
      "id": "observation-gpqa-opus-5-low",
      "fixture": false,
      "modelId": "claude-opus-5-low",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 88.9,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "low effort",
      "scaffold": "No tools",
      "comparability": "comparable"
    },
    {
      "id": "observation-gpqa-sol",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 94.1,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "No tools",
      "comparability": "comparable"
    },
    {
      "id": "observation-gpqa-grok",
      "fixture": false,
      "modelId": "grok-4-6-high",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 94.9,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "No tools",
      "comparability": "comparable"
    },
    {
      "id": "observation-cursorbench-3-2-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 70.8,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 2.81,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 41136,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 46,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 70.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 17.32,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 103525,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 72,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 70,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 8.23,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 61838,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 78,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-grok-4-6-high",
      "fixture": false,
      "modelId": "grok-4-6-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 69.9,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 2.34,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 32449,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 39,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-opus-5-xhigh",
      "fixture": false,
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 69.3,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 7.35,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 54239,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 72,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-fable-5-xhigh",
      "fixture": false,
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 68.4,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 11.73,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 64971,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 56,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 67.2,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 5.69,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 28320,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 48,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-grok-4-6-medium",
      "fixture": false,
      "modelId": "grok-4-6-medium",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 67.1,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "medium effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 1.28,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 17942,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 29,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-opus-5-high",
      "fixture": false,
      "modelId": "claude-opus-5-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 66.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 3.91,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 27932,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 48,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-fable-5-high",
      "fixture": false,
      "modelId": "claude-fable-5-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 66.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 8.77,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 43747,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 48,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-fable-5-medium",
      "fixture": false,
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 65.2,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "medium effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 6.8,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 30366,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 41,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-terra-max",
      "fixture": false,
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 64.9,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-terra",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 2.31,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 32969,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 47,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-sol-xhigh",
      "fixture": false,
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 64.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 3.88,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 19699,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 38,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-opus-5-medium",
      "fixture": false,
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 64.3,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "medium effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 3.29,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 23612,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 44,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-sol-high",
      "fixture": false,
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 63.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 2.79,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 13867,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 32,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-opus-5-low",
      "fixture": false,
      "modelId": "claude-opus-5-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 62.8,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 2.55,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 18529,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 37,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-fable-5-low",
      "fixture": false,
      "modelId": "claude-fable-5-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 62.1,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 4.46,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 18182,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 31,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-sonnet-5-max",
      "fixture": false,
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 61.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-sonnet-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 4.3,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 92882,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 86,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-luna-max",
      "fixture": false,
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 61.1,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-luna",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.39,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 87973,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 61,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-grok-4-6-low",
      "fixture": false,
      "modelId": "grok-4-6-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 61,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.7,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 10658,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 23,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-sol-medium",
      "fixture": false,
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 60,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "medium effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 1.95,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 9747,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 27,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-kimi-k3-high",
      "fixture": false,
      "modelId": "kimi-k3-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 59.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "kimi-k3",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 1.89,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 26846,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 47,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-terra-xhigh",
      "fixture": false,
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 59.2,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-terra",
      "reasoningMode": "reasoning",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 1.15,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 16089,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 29,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gemini-3-7-flash-medium",
      "fixture": false,
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 59,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "thinking",
      "reasoningBudget": "medium effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.95,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 30953,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 82,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-sonnet-5-xhigh",
      "fixture": false,
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 58.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-sonnet-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 2.77,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 52871,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 67,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-luna-xhigh",
      "fixture": false,
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 57.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-luna",
      "reasoningMode": "reasoning",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.23,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 22480,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 48,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-sonnet-5-high",
      "fixture": false,
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 56.9,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-sonnet-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 2.13,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 39483,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 57,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-glm-5-2-max",
      "fixture": false,
      "modelId": "glm-5-2-max",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 55,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "glm-5.2",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 1.76,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 35946,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 58,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-terra-high",
      "fixture": false,
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 54.2,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-terra",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.71,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 9468,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 23,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gemini-3-7-flash-low",
      "fixture": false,
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 53.8,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "thinking",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.74,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 20594,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 68,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-luna-high",
      "fixture": false,
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 56.8,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-luna",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.16,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 15141,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 40,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-sol-low",
      "fixture": false,
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 52.6,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 1.01,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 5104,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 19,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-sonnet-5-medium",
      "fixture": false,
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 52.4,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-sonnet-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "medium effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 1.44,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 26200,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 46,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-glm-5-2-high",
      "fixture": false,
      "modelId": "glm-5-2-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 51.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "glm-5.2",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 1.19,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 21829,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 49,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-kimi-k3-low",
      "fixture": false,
      "modelId": "kimi-k3-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 50.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "kimi-k3",
      "reasoningMode": "reasoning",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.99,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 13007,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 33,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-terra-medium",
      "fixture": false,
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 50.3,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-terra",
      "reasoningMode": "reasoning",
      "reasoningBudget": "medium effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.49,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 6222,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 20,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-luna-medium",
      "fixture": false,
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 47.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-luna",
      "reasoningMode": "reasoning",
      "reasoningBudget": "medium effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.08,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 7095,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 28,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-claude-sonnet-5-low",
      "fixture": false,
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 47.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "claude-sonnet-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.87,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 16269,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 33,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-terra-low",
      "fixture": false,
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 46.9,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-terra",
      "reasoningMode": "reasoning",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.42,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 5312,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 19,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-cursorbench-3-2-gpt-5-6-luna-low",
      "fixture": false,
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 37.6,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "vendor",
      "sourceName": "Cursor CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gpt-5.6-luna",
      "reasoningMode": "reasoning",
      "reasoningBudget": "low effort",
      "scaffold": "Cursor agent workflow",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.03,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Tokens / task",
          "value": 3209,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Steps / task",
          "value": 17,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-deepswe-1-1-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 73.64864864864865,
      "evaluatedAt": "2026-08-13T16:11:55.708Z",
      "firstPublishedAt": "2026-08-13T16:11:55.708Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 live leaderboard artifact",
      "sourceUrl": "https://deepswe.datacurve.ai/artifacts/v1.1/leaderboard-live.json",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "bash"
      ],
      "uncertainty": {
        "ciLower": 69.77633822222292,
        "ciUpper": 77.52095907502037,
        "confidenceLevel": 0.95,
        "runCount": 4
      },
      "secondaryMetrics": [
        {
          "label": "Passed attempts",
          "value": 327,
          "unit": "attempts",
          "precision": 0
        },
        {
          "label": "Scored attempts",
          "value": 444,
          "unit": "attempts",
          "precision": 0
        }
      ],
      "comparability": "comparable",
      "comparabilityNote": "Same DeepSWE v1.1 artifact, mini-swe-agent harness, and pass@1 attempt-rate definition.",
      "notes": "Ingested from the pinned v1.1 live JSON artifact; raw score is pass@1 × 100."
    },
    {
      "id": "observation-deepswe-1-1-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 72.66666666666667,
      "evaluatedAt": "2026-08-13T16:11:55.708Z",
      "firstPublishedAt": "2026-08-13T16:11:55.708Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 live leaderboard artifact",
      "sourceUrl": "https://deepswe.datacurve.ai/artifacts/v1.1/leaderboard-live.json",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "bash"
      ],
      "uncertainty": {
        "ciLower": 69.83684441682949,
        "ciUpper": 75.49648891650386,
        "confidenceLevel": 0.95,
        "runCount": 4
      },
      "secondaryMetrics": [
        {
          "label": "Passed attempts",
          "value": 327,
          "unit": "attempts",
          "precision": 0
        },
        {
          "label": "Scored attempts",
          "value": 450,
          "unit": "attempts",
          "precision": 0
        }
      ],
      "comparability": "comparable",
      "comparabilityNote": "Same DeepSWE v1.1 artifact, mini-swe-agent harness, and pass@1 attempt-rate definition.",
      "notes": "Ingested from the pinned v1.1 live JSON artifact; raw score is pass@1 × 100."
    },
    {
      "id": "observation-deepswe-1-1-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 69.72477064220185,
      "evaluatedAt": "2026-08-13T16:11:55.708Z",
      "firstPublishedAt": "2026-08-13T16:11:55.708Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 live leaderboard artifact",
      "sourceUrl": "https://deepswe.datacurve.ai/artifacts/v1.1/leaderboard-live.json",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "bash"
      ],
      "uncertainty": {
        "ciLower": 65.69129639391394,
        "ciUpper": 73.75824489300977,
        "confidenceLevel": 0.95,
        "runCount": 4
      },
      "secondaryMetrics": [
        {
          "label": "Passed attempts",
          "value": 304,
          "unit": "attempts",
          "precision": 0
        },
        {
          "label": "Scored attempts",
          "value": 436,
          "unit": "attempts",
          "precision": 0
        }
      ],
      "comparability": "comparable",
      "comparabilityNote": "Same DeepSWE v1.1 artifact, mini-swe-agent harness, and pass@1 attempt-rate definition.",
      "notes": "Ingested from the pinned v1.1 live JSON artifact; raw score is pass@1 × 100."
    },
    {
      "id": "observation-deepswe-1-1-gpt-5-6-luna-max",
      "fixture": false,
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 67,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 leaderboard",
      "sourceUrl": "https://deepswe.datacurve.ai/",
      "modelVersion": "gpt-5.6-luna",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 63,
        "ciUpper": 71,
        "confidenceLevel": 0.95
      },
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.61,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Output tokens / task",
          "value": 73000,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Agent steps / task",
          "value": 102,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "comparable"
    },
    {
      "id": "observation-deepswe-1-1-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 66.74057649667405,
      "evaluatedAt": "2026-08-13T16:11:55.708Z",
      "firstPublishedAt": "2026-08-13T16:11:55.708Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 live leaderboard artifact",
      "sourceUrl": "https://deepswe.datacurve.ai/artifacts/v1.1/leaderboard-live.json",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "xhigh effort",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "bash"
      ],
      "uncertainty": {
        "ciLower": 64.56489227221078,
        "ciUpper": 68.91626072107911,
        "confidenceLevel": 0.95,
        "runCount": 4
      },
      "secondaryMetrics": [
        {
          "label": "Passed attempts",
          "value": 301,
          "unit": "attempts",
          "precision": 0
        },
        {
          "label": "Scored attempts",
          "value": 451,
          "unit": "attempts",
          "precision": 0
        }
      ],
      "comparability": "comparable",
      "comparabilityNote": "Same DeepSWE v1.1 artifact, mini-swe-agent harness, and pass@1 attempt-rate definition.",
      "notes": "Ingested from the pinned v1.1 live JSON artifact; raw score is pass@1 × 100."
    },
    {
      "id": "observation-deepswe-1-1-claude-sonnet-5-max",
      "fixture": false,
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 54,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 leaderboard",
      "sourceUrl": "https://deepswe.datacurve.ai/",
      "modelVersion": "claude-sonnet-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 50,
        "ciUpper": 58,
        "confidenceLevel": 0.95
      },
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 26.4,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Output tokens / task",
          "value": 214000,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Agent steps / task",
          "value": 268,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "comparable"
    },
    {
      "id": "observation-deepswe-1-1-deepseek-v4-flash-max",
      "fixture": false,
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 53,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 leaderboard",
      "sourceUrl": "https://deepswe.datacurve.ai/",
      "modelVersion": "deepseek-v4-flash",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 49,
        "ciUpper": 57,
        "confidenceLevel": 0.95
      },
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 0.1,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Output tokens / task",
          "value": 108000,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Agent steps / task",
          "value": 153,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "comparable"
    },
    {
      "id": "observation-deepswe-1-1-glm-5-2-max",
      "fixture": false,
      "modelId": "glm-5-2-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 44,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 leaderboard",
      "sourceUrl": "https://deepswe.datacurve.ai/",
      "modelVersion": "glm-5.2",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 42,
        "ciUpper": 46,
        "confidenceLevel": 0.95
      },
      "secondaryMetrics": [
        {
          "label": "Cost / task",
          "value": 3.92,
          "unit": "USD",
          "precision": 2
        },
        {
          "label": "Output tokens / task",
          "value": 78000,
          "unit": "tokens",
          "precision": 0
        },
        {
          "label": "Agent steps / task",
          "value": 129,
          "unit": "steps",
          "precision": 0
        }
      ],
      "comparability": "comparable"
    },
    {
      "id": "observation-apex-agents-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "apex-agents",
      "benchmarkVersion": "APEX-Agents public leaderboard snapshot",
      "status": "available",
      "rawScore": 60.6,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-agents",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-Agents leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-agents-leaderboard/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Mercor Archipelago",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 57,
        "ciUpper": 64.2,
        "confidenceLevel": 0.95
      },
      "secondaryMetrics": [
        {
          "label": "Pass@1",
          "value": 43.5,
          "unit": "%",
          "precision": 1
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability.",
      "notes": "Canonical metric is mean rubric score; Pass@1 is retained as a separate secondary metric and is not substituted."
    },
    {
      "id": "observation-apex-agents-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "apex-agents",
      "benchmarkVersion": "APEX-Agents public leaderboard snapshot",
      "status": "available",
      "rawScore": 59.2,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-agents",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-Agents leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-agents-leaderboard/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Mercor Archipelago",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 55.5,
        "ciUpper": 62.900000000000006,
        "confidenceLevel": 0.95
      },
      "secondaryMetrics": [
        {
          "label": "Pass@1",
          "value": 43.3,
          "unit": "%",
          "precision": 1
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability.",
      "notes": "Canonical metric is mean rubric score; Pass@1 is retained as a separate secondary metric and is not substituted."
    },
    {
      "id": "observation-apex-agents-grok-4-6-high",
      "fixture": false,
      "modelId": "grok-4-6-high",
      "benchmarkId": "apex-agents",
      "benchmarkVersion": "APEX-Agents public leaderboard snapshot",
      "status": "available",
      "rawScore": 57.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-agents",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-Agents leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-agents-leaderboard/",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Mercor Archipelago",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 54,
        "ciUpper": 61,
        "confidenceLevel": 0.95
      },
      "secondaryMetrics": [
        {
          "label": "Pass@1",
          "value": 41.2,
          "unit": "%",
          "precision": 1
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability.",
      "notes": "Canonical metric is mean rubric score; Pass@1 is retained as a separate secondary metric and is not substituted."
    },
    {
      "id": "observation-apex-agents-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "apex-agents",
      "benchmarkVersion": "APEX-Agents public leaderboard snapshot",
      "status": "available",
      "rawScore": 56.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-agents",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-Agents leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-agents-leaderboard/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Mercor Archipelago",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 53.400000000000006,
        "ciUpper": 60,
        "confidenceLevel": 0.95
      },
      "secondaryMetrics": [
        {
          "label": "Pass@1",
          "value": 39.9,
          "unit": "%",
          "precision": 1
        }
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability.",
      "notes": "Canonical metric is mean rubric score; Pass@1 is retained as a separate secondary metric and is not substituted."
    },
    {
      "id": "observation-apex-swe-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "apex-swe",
      "benchmarkVersion": "APEX-SWE public leaderboard snapshot",
      "status": "available",
      "rawScore": 63.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Mercor Archipelago",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 57.300000000000004,
        "ciUpper": 70.10000000000001,
        "confidenceLevel": 0.95
      },
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-apex-swe-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "apex-swe",
      "benchmarkVersion": "APEX-SWE public leaderboard snapshot",
      "status": "available",
      "rawScore": 58.8,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Mercor Archipelago",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 52.4,
        "ciUpper": 65.2,
        "confidenceLevel": 0.95
      },
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-apex-swe-grok-4-6-high",
      "fixture": false,
      "modelId": "grok-4-6-high",
      "benchmarkId": "apex-swe",
      "benchmarkVersion": "APEX-SWE public leaderboard snapshot",
      "status": "available",
      "rawScore": 56.4,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/",
      "modelVersion": "grok-4.6",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Mercor Archipelago",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "uncertainty": {
        "ciLower": 50.199999999999996,
        "ciUpper": 62.6,
        "confidenceLevel": 0.95
      },
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability."
    },
    {
      "id": "observation-frontiercode-claude-fable-5-low",
      "fixture": false,
      "modelId": "claude-fable-5-low",
      "benchmarkId": "frontiercode",
      "benchmarkVersion": "FrontierCode v1.1 Extended",
      "status": "available",
      "rawScore": 60.8,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-frontiercode-11",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition FrontierCode result table",
      "sourceUrl": "https://cognition.ai/blog/frontier-code",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "low effort",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability.",
      "notes": "Published as the Fable 5 low-effort Extended Score; no sidekick configuration is inferred."
    },
    {
      "id": "observation-harvey-lab-aa-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "harvey-lab-aa",
      "benchmarkVersion": "Harvey LAB-AA public leaderboard snapshot",
      "status": "available",
      "rawScore": 94.6,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-harvey-lab-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Harvey LAB-AA leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/harvey-lab-aa",
      "modelVersion": "kimi-k3",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Artificial Analysis Stirrup",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "comparable"
    },
    {
      "id": "observation-harvey-lab-aa-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "harvey-lab-aa",
      "benchmarkVersion": "Harvey LAB-AA public leaderboard snapshot",
      "status": "available",
      "rawScore": 93.6,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-harvey-lab-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Harvey LAB-AA leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/harvey-lab-aa",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Artificial Analysis Stirrup",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability.",
      "notes": "Source identifies the Opus 4.8 fallback behavior for the Fable configuration."
    },
    {
      "id": "observation-aa-briefcase-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "aa-briefcase",
      "benchmarkVersion": "AA-Briefcase public leaderboard snapshot",
      "status": "available",
      "rawScore": 1587,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-briefcase-secondary",
      "sourceType": "third_party",
      "sourceName": "eesel AI AA-Briefcase launch summary",
      "sourceUrl": "https://www.eesel.ai/blog/aa-briefcase",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "adaptive thinking",
      "reasoningBudget": "max effort",
      "scaffold": "Artificial Analysis Stirrup",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Source leaderboard uses a benchmark-maintainer or vendor-specific agent harness; score is retained with limited cross-harness comparability.",
      "notes": "Dated launch-era secondary-source Elo; the live Artificial Analysis page remains the canonical benchmark definition."
    },
    {
      "id": "observation-deepswe-1-1-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 11.72566371681416,
      "evaluatedAt": "2026-08-13T16:11:55.708Z",
      "firstPublishedAt": "2026-08-13T16:11:55.708Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepswe-11",
      "sourceType": "benchmark_org",
      "sourceName": "Datacurve DeepSWE v1.1 live leaderboard artifact",
      "sourceUrl": "https://deepswe.datacurve.ai/artifacts/v1.1/leaderboard-live.json",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high",
      "scaffold": "mini-swe-agent",
      "toolAccess": [
        "bash"
      ],
      "uncertainty": {
        "ciLower": 10.2445682557053,
        "ciUpper": 13.20675911790715,
        "confidenceLevel": 0.95,
        "runCount": 4
      },
      "secondaryMetrics": [
        {
          "label": "Passed attempts",
          "value": 53,
          "unit": "attempts",
          "precision": 0
        },
        {
          "label": "Scored attempts",
          "value": 452,
          "unit": "attempts",
          "precision": 0
        }
      ],
      "comparability": "comparable",
      "comparabilityNote": "Same DeepSWE v1.1 artifact, mini-swe-agent harness, and pass@1 attempt-rate definition.",
      "notes": "Ingested from the pinned v1.1 live JSON artifact; raw score is pass@1 × 100."
    },
    {
      "id": "observation-gpqa-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 92.6,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "No tools",
      "comparability": "comparable",
      "comparabilityNote": "Public row includes an Opus 4.8 fallback for a subset of prompts; retained as caveated evidence.",
      "notes": "Live GPQA Diamond row reconciled on the dataset snapshot date."
    },
    {
      "id": "observation-gpqa-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 94.1,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-aa-gpqa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "reasoning",
      "reasoningBudget": "high",
      "scaffold": "No tools",
      "comparability": "comparable",
      "notes": "Live GPQA Diamond row reconciled on the dataset snapshot date."
    },
    {
      "id": "observation-code-migration-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "code-migration",
      "benchmarkVersion": "Vals Code Migration · CLI split · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 53.3,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-code-migration",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Code Migration leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/code-migration",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Code Migration harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-code-migration-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "code-migration",
      "benchmarkVersion": "Vals Code Migration · CLI split · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 60.1,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-code-migration",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Code Migration leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/code-migration",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Code Migration harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-code-migration-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "code-migration",
      "benchmarkVersion": "Vals Code Migration · CLI split · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 47.2,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-code-migration",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Code Migration leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/code-migration",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Code Migration harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-code-migration-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "code-migration",
      "benchmarkVersion": "Vals Code Migration · CLI split · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 39.4,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-code-migration",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Code Migration leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/code-migration",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Code Migration harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-code-migration-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "code-migration",
      "benchmarkVersion": "Vals Code Migration · CLI split · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 10.8,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-code-migration",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Code Migration leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/code-migration",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Code Migration harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-code-migration-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "code-migration",
      "benchmarkVersion": "Vals Code Migration · CLI split · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 1.5,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-code-migration",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Code Migration leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/code-migration",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Code Migration harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "The published CLI table records substantial timeout losses; this is retained as a benchmark result rather than silently converted to a capability-only score."
    },
    {
      "id": "observation-code-migration-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "code-migration",
      "benchmarkVersion": "Vals Code Migration · CLI split · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 32.1,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-code-migration",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Code Migration leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/code-migration",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Code Migration harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-excel-modeling-benchmark-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "excel-modeling-benchmark",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 88,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-excel-modeling",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Excel Modeling Benchmark leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/emb",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Excel Modeling harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-excel-modeling-benchmark-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "excel-modeling-benchmark",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 85,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-excel-modeling",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Excel Modeling Benchmark leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/emb",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Excel Modeling harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-excel-modeling-benchmark-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "excel-modeling-benchmark",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 85,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-excel-modeling",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Excel Modeling Benchmark leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/emb",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Excel Modeling harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-excel-modeling-benchmark-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 78,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-excel-modeling",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Excel Modeling Benchmark leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/emb",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Excel Modeling harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-excel-modeling-benchmark-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "excel-modeling-benchmark",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 79,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-excel-modeling",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Excel Modeling Benchmark leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/emb",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Excel Modeling harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-excel-modeling-benchmark-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "excel-modeling-benchmark",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 85,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-excel-modeling",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Excel Modeling Benchmark leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/emb",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Excel Modeling harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-excel-modeling-benchmark-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "excel-modeling-benchmark",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 67,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-excel-modeling",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Excel Modeling Benchmark leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/emb",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Excel Modeling harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-legal-research-bench-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "legal-research-bench",
      "benchmarkVersion": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 77,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-legal-research",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Legal Research Bench leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/legal_research",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Legal Research harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-legal-research-bench-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "legal-research-bench",
      "benchmarkVersion": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 67,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-legal-research",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Legal Research Bench leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/legal_research",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Legal Research harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-legal-research-bench-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "legal-research-bench",
      "benchmarkVersion": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 80,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-legal-research",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Legal Research Bench leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/legal_research",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Legal Research harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-legal-research-bench-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "legal-research-bench",
      "benchmarkVersion": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 73,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-legal-research",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Legal Research Bench leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/legal_research",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Legal Research harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-legal-research-bench-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "legal-research-bench",
      "benchmarkVersion": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 33,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-legal-research",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Legal Research Bench leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/legal_research",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Legal Research harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-legal-research-bench-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "legal-research-bench",
      "benchmarkVersion": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 63,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-legal-research",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Legal Research Bench leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/legal_research",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Legal Research harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-legal-research-bench-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "legal-research-bench",
      "benchmarkVersion": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot",
      "status": "available",
      "rawScore": 60,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-legal-research",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Legal Research Bench leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/legal_research",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Legal Research harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-harvey-lab-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "harvey-lab-vals",
      "benchmarkVersion": "Vals Harvey LAB · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 11.25,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-harvey-lab",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Harvey Legal Agent Benchmark · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/hlab",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Harvey LAB harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "The Vals page reports fallback to Claude Opus 4.8 on four tasks; the published result is retained with the fallback caveat and is not a pure Fable-only run."
    },
    {
      "id": "observation-harvey-lab-vals-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "benchmarkVersion": "Vals Harvey LAB · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 15.83,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-harvey-lab",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Harvey Legal Agent Benchmark · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/hlab",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Harvey LAB harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-vibe-code-bench-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "vibe-code-bench",
      "benchmarkVersion": "Vals Vibe Code Bench v1.1 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 88.4,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-vibe-code",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Vibe Code Bench v1.1 leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/vibe-code",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Vibe Code Bench v1.1 harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-vibe-code-bench-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "vibe-code-bench",
      "benchmarkVersion": "Vals Vibe Code Bench v1.1 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 90.35,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-vibe-code",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI Vibe Code Bench v1.1 leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/vibe-code",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI Vibe Code Bench v1.1 harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-mmlu-pro-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "mmlu-pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 90.99,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-mmlu-pro",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI MMLU-Pro · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/mmlu_pro",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI MMLU-Pro evaluation harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-mmmu-pro-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "mmmu-pro",
      "benchmarkVersion": "Vals MMMU-Pro · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 89.88,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-mmmu-pro",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI MMMU-Pro · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/mmmu",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI MMMU-Pro evaluation harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-mmmu-pro-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "mmmu-pro",
      "benchmarkVersion": "Vals MMMU-Pro · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 89.31,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-mmmu-pro",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI MMMU-Pro · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/mmmu",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI MMMU-Pro evaluation harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-mmmu-pro-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "mmmu-pro",
      "benchmarkVersion": "Vals MMMU-Pro · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 88.84,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-mmmu-pro",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI MMMU-Pro · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/mmmu",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI MMMU-Pro evaluation harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-livecodebench-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "livecodebench",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 89.03,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-livecodebench",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI LiveCodeBench · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/lcb",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI LiveCodeBench harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-livecodebench-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "livecodebench",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 89.78,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-livecodebench",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI LiveCodeBench · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/lcb",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI LiveCodeBench harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-livecodebench-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "livecodebench",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 82.6,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-livecodebench",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI LiveCodeBench · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/lcb",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI LiveCodeBench harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-livecodebench-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "livecodebench",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 88.48,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-livecodebench",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI LiveCodeBench · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/lcb",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI LiveCodeBench harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-programbench-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "programbench",
      "benchmarkVersion": "Vals ProgramBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 82.3,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-programbench",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI ProgramBench · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/programbench",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI ProgramBench harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-programbench-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "programbench",
      "benchmarkVersion": "Vals ProgramBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 76.8,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-programbench",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI ProgramBench · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/programbench",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI ProgramBench harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "The Vals page reports fallback behavior for the Fable 5 run; this value is displayed with that caveat and is not treated as a pure Fable-only evaluation."
    },
    {
      "id": "observation-programbench-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "programbench",
      "benchmarkVersion": "Vals ProgramBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 77.6,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-programbench",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI ProgramBench · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/programbench",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI ProgramBench harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-programbench-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "programbench",
      "benchmarkVersion": "Vals ProgramBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 62.77,
      "evaluatedAt": "2026-08-13T00:00:00.000Z",
      "firstPublishedAt": "2026-08-13T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-vals-programbench",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI ProgramBench · Vals run leaderboard",
      "sourceUrl": "https://vals.ai/benchmarks/programbench",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Vals AI provider-default configuration",
      "scaffold": "Vals AI ProgramBench harness",
      "comparability": "comparable",
      "comparabilityNote": "Rows share the Vals benchmark source and scaffold, but the public page does not consistently pin the requested frontier reasoning effort.",
      "notes": "Directly readable public row retained as displayed evidence. The source does not establish the requested max/xhigh effort for this model."
    },
    {
      "id": "observation-apex-swe-integration-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "apex-swe-integration",
      "benchmarkVersion": "APEX-SWE Integration track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 64,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-integration",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Integration leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/integration-swe/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-integration-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "apex-swe-integration",
      "benchmarkVersion": "APEX-SWE Integration track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 63.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-integration",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Integration leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/integration-swe/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-integration-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "apex-swe-integration",
      "benchmarkVersion": "APEX-SWE Integration track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 55.7,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-integration",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Integration leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/integration-swe/",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-integration-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "apex-swe-integration",
      "benchmarkVersion": "APEX-SWE Integration track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 60,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-integration",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Integration leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/integration-swe/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-integration-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "apex-swe-integration",
      "benchmarkVersion": "APEX-SWE Integration track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 50.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-integration",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Integration leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/integration-swe/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-observability-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "apex-swe-observability",
      "benchmarkVersion": "APEX-SWE Observability track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 63.5,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-observability",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Observability leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/observability-swe/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-observability-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "apex-swe-observability",
      "benchmarkVersion": "APEX-SWE Observability track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 54.2,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-observability",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Observability leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/observability-swe/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-observability-gemini-3-1-pro-high",
      "fixture": false,
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "apex-swe-observability",
      "benchmarkVersion": "APEX-SWE Observability track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 12,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-observability",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Observability leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/observability-swe/",
      "modelVersion": "gemini-3.1-pro-preview",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-observability-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "apex-swe-observability",
      "benchmarkVersion": "APEX-SWE Observability track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 36,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-observability",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Observability leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/observability-swe/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-apex-swe-observability-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "apex-swe-observability",
      "benchmarkVersion": "APEX-SWE Observability track · public leaderboard snapshot",
      "status": "available",
      "rawScore": 17,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-apex-swe-observability",
      "sourceType": "benchmark_org",
      "sourceName": "Mercor APEX-SWE · Observability leaderboard",
      "sourceUrl": "https://www.mercor.com/apex/apex-swe-leaderboard/observability-swe/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "agent reasoning",
      "reasoningBudget": "max effort",
      "scaffold": "Terminus-2",
      "toolAccess": [
        "browser",
        "shell"
      ],
      "comparability": "limited",
      "comparabilityNote": "Track rows use the public Terminus-2 label; the existing aggregate APEX-SWE record uses Mercor Archipelago, so this track remains outside the score until the scaffold identity is reconciled.",
      "notes": "Exact target model and effort row from the track-specific public leaderboard."
    },
    {
      "id": "observation-aa-omniscience-index-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "aa-omniscience-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 1,
      "evaluatedAt": "2026-08-15T00:00:00.000Z",
      "firstPublishedAt": "2026-08-15T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-15T00:00:00.000Z",
      "citationId": "citation-deepseek-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis DeepSeek V4 Pro 0813 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/deepseek-v4-pro",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-swe-bench-verified-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "swe-bench-verified",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 95,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI mini-SWE-agent bash harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI mini-SWE-agent bash harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-swe-bench-verified-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "swe-bench-verified",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 96.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI mini-SWE-agent bash harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI mini-SWE-agent bash harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-swe-bench-verified-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "swe-bench-verified",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 97,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI mini-SWE-agent bash harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI mini-SWE-agent bash harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-swe-bench-verified-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "swe-bench-verified",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 93.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI mini-SWE-agent bash harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI mini-SWE-agent bash harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-swe-bench-verified-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "swe-bench-verified",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 95.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI mini-SWE-agent bash harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI mini-SWE-agent bash harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-swe-bench-verified-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "swe-bench-verified",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 85.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI mini-SWE-agent bash harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI mini-SWE-agent bash harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-swe-bench-verified-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "swe-bench-verified",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 96.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI mini-SWE-agent bash harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI mini-SWE-agent bash harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-swe-bench-verified-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "swe-bench-verified",
      "benchmarkVersion": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 86.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI mini-SWE-agent bash harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI mini-SWE-agent bash harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-exploitbench-v8-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "exploitbench-v8",
      "benchmarkVersion": "ExploitBench V8 · AutoNudge arm · 2026-07-24",
      "status": "available",
      "rawScore": 70,
      "evaluatedAt": "2026-07-24T00:00:00.000Z",
      "firstPublishedAt": "2026-07-24T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-anthropic-opus-5-system-card",
      "sourceType": "vendor",
      "sourceName": "Anthropic",
      "sourceUrl": "https://www-cdn.anthropic.com/b514064af1408018e64b1ad24e7d5e75850b4ffd/Claude%20Opus%205%20System%20Card.pdf",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "ExploitBench authors’ harness with Anthropic timeout overlay",
      "comparability": "limited",
      "comparabilityNote": "Exact Opus 5 AutoNudge Cap% result; cross-provider sheet values may use independently reported harness configurations.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-matharena-arxivmath-2026-05-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "benchmarkVersion": "MathArena ArXivMath · 05/2026",
      "status": "available",
      "rawScore": 87.5,
      "evaluatedAt": "2026-05-31T00:00:00.000Z",
      "firstPublishedAt": "2026-05-31T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-fable-5-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/anthropic_fable_5_max",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena May 2026 competition harness",
      "uncertainty": {
        "ciLower": 81.58,
        "ciUpper": 93.42,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Exact evaluator-hosted May 2026 competition result.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-arxivmath-2026-05-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "benchmarkVersion": "MathArena ArXivMath · 05/2026",
      "status": "available",
      "rawScore": 61.67,
      "evaluatedAt": "2026-05-31T00:00:00.000Z",
      "firstPublishedAt": "2026-05-31T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-kimi-k3",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/moonshot_k3",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena May 2026 competition harness",
      "uncertainty": {
        "ciLower": 52.97,
        "ciUpper": 70.37,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Exact evaluator-hosted May 2026 competition result.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-arxivmath-2026-06-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "benchmarkVersion": "MathArena ArXivMath · 06/2026",
      "status": "available",
      "rawScore": 83.67,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-fable-5-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/anthropic_fable_5_max",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 77.69,
        "ciUpper": 89.65,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-arxivmath-2026-06-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "benchmarkVersion": "MathArena ArXivMath · 06/2026",
      "status": "available",
      "rawScore": 86.73,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-sol-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/openai_gpt_56_sol",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 80.01,
        "ciUpper": 93.45,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-arxivmath-2026-06-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "benchmarkVersion": "MathArena ArXivMath · 06/2026",
      "status": "available",
      "rawScore": 80.95,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-opus-5-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/anthropic_opus_5_max",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 74.6,
        "ciUpper": 87.3,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-arxivmath-2026-06-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "benchmarkVersion": "MathArena ArXivMath · 06/2026",
      "status": "available",
      "rawScore": 72.11,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-kimi-k3",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/moonshot_k3",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 64.86,
        "ciUpper": 79.36,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-brokenarxiv-2026-06-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "benchmarkVersion": "MathArena BrokenArXiv · 06/2026",
      "status": "available",
      "rawScore": 47.84,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-fable-5-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/anthropic_fable_5_max",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 40.15,
        "ciUpper": 55.53,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-brokenarxiv-2026-06-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "benchmarkVersion": "MathArena BrokenArXiv · 06/2026",
      "status": "available",
      "rawScore": 67.28,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-sol-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/openai_gpt_56_sol",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 60.06,
        "ciUpper": 74.5,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-brokenarxiv-2026-06-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "benchmarkVersion": "MathArena BrokenArXiv · 06/2026",
      "status": "available",
      "rawScore": 90.74,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-opus-5-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/anthropic_opus_5_max",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 85.27,
        "ciUpper": 96.21,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-brokenarxiv-2026-06-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "benchmarkVersion": "MathArena BrokenArXiv · 06/2026",
      "status": "available",
      "rawScore": 51.85,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-kimi-k3",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/moonshot_k3",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 42.43,
        "ciUpper": 61.27,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-arxivlean-2026-06-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "benchmarkVersion": "MathArena ArXivLean · 06/2026",
      "status": "available",
      "rawScore": 14.58,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-fable-5-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/anthropic_fable_5_max",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 4.6,
        "ciUpper": 24.56,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-arxivlean-2026-06-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "benchmarkVersion": "MathArena ArXivLean · 06/2026",
      "status": "available",
      "rawScore": 37.5,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-sol-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/openai_gpt_56_sol",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 23.8,
        "ciUpper": 51.2,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-matharena-arxivlean-2026-06-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "benchmarkVersion": "MathArena ArXivLean · 06/2026",
      "status": "available",
      "rawScore": 31.25,
      "evaluatedAt": "2026-06-30T00:00:00.000Z",
      "firstPublishedAt": "2026-06-30T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-matharena-opus-5-max",
      "sourceType": "benchmark_org",
      "sourceName": "MathArena",
      "sourceUrl": "https://matharena.ai/models/anthropic_opus_5_max",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "MathArena June 2026 competition harness",
      "uncertainty": {
        "ciLower": 18.14,
        "ciUpper": 44.36,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Same evaluator, competition month, metric, and max model configuration.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-scicode-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 60,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-together-frontier-comparison",
      "sourceType": "third_party",
      "sourceName": "Together AI",
      "sourceUrl": "https://www.together.ai/models/glm-52",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-scicode-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 56,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-together-frontier-comparison",
      "sourceType": "third_party",
      "sourceName": "Together AI",
      "sourceUrl": "https://www.together.ai/models/glm-52",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-scicode-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 56,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-together-frontier-comparison",
      "sourceType": "third_party",
      "sourceName": "Together AI",
      "sourceUrl": "https://www.together.ai/models/glm-52",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-scicode-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 58.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-scicode-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 53.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-scicode-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 52.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, OpenLM.ai, and Artificial Analysis",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-scicode-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 49.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-v4-pro-0813",
      "sourceType": "vendor",
      "sourceName": "DeepSeek",
      "sourceUrl": "https://api-docs.deepseek.com/news/news260813/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-scicode-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 56.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-scicode-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "scicode",
      "benchmarkVersion": "SciCode · Together AI comparison snapshot",
      "status": "available",
      "rawScore": 56.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Same comparison table, but underlying evaluator version, harness, and effort settings are not fully disclosed.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-frontiercode-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "frontiercode",
      "benchmarkVersion": "FrontierCode v1.1 Extended",
      "status": "available",
      "rawScore": 63.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-frontiercode-11-data",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition",
      "sourceUrl": "https://cognition.com/data/frontiercode-leaderboard/data.json",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same subset and metric, with source-disclosed effort differences: Opus uses medium and Grok uses high; Kimi is labeled with no explicit effort.",
      "notes": "Cognition Extended weighted score at max effort."
    },
    {
      "id": "observation-frontiercode-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "frontiercode",
      "benchmarkVersion": "FrontierCode v1.1 Extended",
      "status": "available",
      "rawScore": 60.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-frontiercode-11-data",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition",
      "sourceUrl": "https://cognition.com/data/frontiercode-leaderboard/data.json",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same subset and metric, with source-disclosed effort differences: Opus uses medium and Grok uses high; Kimi is labeled with no explicit effort.",
      "notes": "Cognition Extended weighted score at max effort."
    },
    {
      "id": "observation-frontiercode-claude-opus-5-medium",
      "fixture": false,
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "frontiercode",
      "benchmarkVersion": "FrontierCode v1.1 Extended",
      "status": "available",
      "rawScore": 63.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-frontiercode-11-data",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition",
      "sourceUrl": "https://cognition.com/data/frontiercode-leaderboard/data.json",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same subset and metric, with source-disclosed effort differences: Opus uses medium and Grok uses high; Kimi is labeled with no explicit effort.",
      "notes": "The sheet’s Opus column is redirected to the exact Opus 5 Medium configuration reported by Cognition."
    },
    {
      "id": "observation-frontiercode-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "frontiercode",
      "benchmarkVersion": "FrontierCode v1.1 Extended",
      "status": "available",
      "rawScore": 58.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-frontiercode-11-data",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition",
      "sourceUrl": "https://cognition.com/data/frontiercode-leaderboard/data.json",
      "modelVersion": "kimi-k3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same subset and metric, with source-disclosed effort differences: Opus uses medium and Grok uses high; Kimi is labeled with no explicit effort.",
      "notes": "Cognition reports Kimi K3 at 58.19 weighted score and 63.58 pass rate; its effort field is none, so this display-only observation is not treated as effort-pinned."
    },
    {
      "id": "observation-frontiercode-grok-4-6-high",
      "fixture": false,
      "modelId": "grok-4-6-high",
      "benchmarkId": "frontiercode",
      "benchmarkVersion": "FrontierCode v1.1 Extended",
      "status": "available",
      "rawScore": 61.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-frontiercode-11-data",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition",
      "sourceUrl": "https://cognition.com/data/frontiercode-leaderboard/data.json",
      "modelVersion": "grok-4.6",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same subset and metric, with source-disclosed effort differences: Opus uses medium and Grok uses high; Kimi is labeled with no explicit effort.",
      "notes": "The sheet’s Grok column is redirected to the exact Grok 4.6 High configuration reported by Cognition."
    },
    {
      "id": "observation-frontiercode-1-1-main-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "frontiercode-1-1-main",
      "benchmarkVersion": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 53.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-frontiercode-11",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition",
      "sourceUrl": "https://cognition.ai/blog/frontier-code",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark version, Main subset, and score metric, but model harnesses and provenance differ across public records.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-frontiercode-1-1-main-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "frontiercode-1-1-main",
      "benchmarkVersion": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 47.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-frontiercode-11",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition",
      "sourceUrl": "https://cognition.ai/blog/frontier-code",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark version, Main subset, and score metric, but model harnesses and provenance differ across public records.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-frontiercode-1-1-main-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "frontiercode-1-1-main",
      "benchmarkVersion": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 53.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-frontiercode-11",
      "sourceType": "benchmark_org",
      "sourceName": "Cognition",
      "sourceUrl": "https://cognition.ai/blog/frontier-code",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark version, Main subset, and score metric, but model harnesses and provenance differ across public records.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-frontiercode-1-1-main-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "frontiercode-1-1-main",
      "benchmarkVersion": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 44.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-kimi-k3-frontiercode-main",
      "sourceType": "third_party",
      "sourceName": "Together AI",
      "sourceUrl": "https://www.together.ai/models/kimi-k3",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 maximum thinking effort",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark version, Main subset, and score metric, but model harnesses and provenance differ across public records.",
      "notes": "Together AI reports 44.2% in a table whose peer scores match the current FrontierCode v1.1 Main leaderboard."
    },
    {
      "id": "observation-frontiercode-1-1-main-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "benchmarkVersion": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 48,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-frontiercode-main",
      "sourceType": "third_party",
      "sourceName": "Digital Applied",
      "sourceUrl": "https://www.digitalapplied.com/blog/vendor-benchmark-tables-reading-disclosed-losses-2026",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark version, Main subset, and score metric, but model harnesses and provenance differ across public records.",
      "notes": "Cognition Main-board result for Grok 4.6 High; the distinct 61.3% Extended result is not substituted."
    },
    {
      "id": "observation-frontiercode-1-1-main-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "frontiercode-1-1-main",
      "benchmarkVersion": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 17.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-frontiercode-main",
      "sourceType": "third_party",
      "sourceName": "LLM Stats",
      "sourceUrl": "https://llm-stats.com/benchmarks/frontiercode-1.1",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro maximum reasoning effort",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark version, Main subset, and score metric, but model harnesses and provenance differ across public records.",
      "notes": "Exact Max-variant result from a secondary leaderboard mirror that marks the record unverified."
    },
    {
      "id": "observation-frontiercode-1-1-main-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "frontiercode-1-1-main",
      "benchmarkVersion": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 43.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-frontiercode-main",
      "sourceType": "vendor",
      "sourceName": "Google DeepMind",
      "sourceUrl": "https://deepmind.google/models/model-cards/gemini-3-7-flash/",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark version, Main subset, and score metric, but model harnesses and provenance differ across public records.",
      "notes": "Google-reported FrontierCode 1.1 Main result from the Gemini 3.7 Flash model card."
    },
    {
      "id": "observation-terminal-bench-2-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 84.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-terminal-bench-2-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 89.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-terminal-bench-2-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 86.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-terminal-bench-2-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 88.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-terminal-bench-2-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 88.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-terminal-bench-2-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 86.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-terminal-bench-2-glm-5-3-max",
      "fixture": false,
      "modelId": "glm-5-3-max",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 88.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "glm-5.3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Asterisked value: Z.ai provider run in Claude Code 2.1.207; the evaluation setup may differ from the other dagger-marked Terminal-Bench 2.1 results."
    },
    {
      "id": "observation-terminal-bench-2-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 78.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/terminalbench-v2-1",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Artificial Analysis evaluation value; DeepSeek also reports a higher 87.9% vendor result under a different harness, which is not used for this dagger row."
    },
    {
      "id": "observation-terminal-bench-2-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 82.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-code",
      "sourceType": "vendor",
      "sourceName": "BenchLM",
      "sourceUrl": "https://benchlm.ai/compare/muse-spark-1-1-vs-muse-spark-1-2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Asterisked value: Muse Spark 1.2 running with Muse Code; this is a model-plus-agent result rather than a bare-model comparison."
    },
    {
      "id": "observation-terminal-bench-2-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "terminal-bench-2",
      "benchmarkVersion": "Terminal-Bench v2.1 · cross-source snapshot",
      "status": "available",
      "rawScore": 85.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-google",
      "sourceType": "vendor",
      "sourceName": "Google DeepMind",
      "sourceUrl": "https://deepmind.google/models/gemini/flash/",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Google-reported Gemini 3.7 Flash high-reasoning result; this is retained for the dagger row without treating it as an Anthropic H2H run."
    },
    {
      "id": "observation-sage-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "sage-vals",
      "benchmarkVersion": "SAGE · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 51.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-sage-vals-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "sage-vals",
      "benchmarkVersion": "SAGE · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 52.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-sage-vals-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "sage-vals",
      "benchmarkVersion": "SAGE · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 49.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-sage-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "sage-vals",
      "benchmarkVersion": "SAGE · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 54.26,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-sage-vals-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "sage-vals",
      "benchmarkVersion": "SAGE · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 28.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-sage-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "sage-vals",
      "benchmarkVersion": "SAGE · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 51.25,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-sage-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "sage-vals",
      "benchmarkVersion": "SAGE · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 47.66,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-corpfin-v2-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "corpfin-v2-vals",
      "benchmarkVersion": "CorpFin v2 · Vals archived snapshot",
      "status": "available",
      "rawScore": 71.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-corpfin-v2-vals-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "corpfin-v2-vals",
      "benchmarkVersion": "CorpFin v2 · Vals archived snapshot",
      "status": "available",
      "rawScore": 64.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-corpfin-v2-vals-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "corpfin-v2-vals",
      "benchmarkVersion": "CorpFin v2 · Vals archived snapshot",
      "status": "available",
      "rawScore": 73.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-corpfin-v2-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "corpfin-v2-vals",
      "benchmarkVersion": "CorpFin v2 · Vals archived snapshot",
      "status": "available",
      "rawScore": 71.56,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-corpfin-v2-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "benchmarkVersion": "CorpFin v2 · Vals archived snapshot",
      "status": "available",
      "rawScore": 65.85,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-corpfin-v2-vals-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "corpfin-v2-vals",
      "benchmarkVersion": "CorpFin v2 · Vals archived snapshot",
      "status": "available",
      "rawScore": 65.42,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-corpfin-v2-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "benchmarkVersion": "CorpFin v2 · Vals archived snapshot",
      "status": "available",
      "rawScore": 70.94,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mortgage-tax-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "mortgage-tax-vals",
      "benchmarkVersion": "MortgageTax · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 68.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mortgage-tax-vals-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "mortgage-tax-vals",
      "benchmarkVersion": "MortgageTax · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 67.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mortgage-tax-vals-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "mortgage-tax-vals",
      "benchmarkVersion": "MortgageTax · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 72.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mortgage-tax-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "mortgage-tax-vals",
      "benchmarkVersion": "MortgageTax · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 66.34,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mortgage-tax-vals-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "benchmarkVersion": "MortgageTax · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 64.19,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mortgage-tax-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "benchmarkVersion": "MortgageTax · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 63.99,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mortgage-tax-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "benchmarkVersion": "MortgageTax · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 65.42,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 56.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 53.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 58.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 54.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 53.68,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 50.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 50.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 60.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-finance-agent-v2-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "finance-agent-v2",
      "benchmarkVersion": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 59.55,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/google_gemini-3.7-flash",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Vals AI Finance Agent v2 harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Finance Agent v2 harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 73.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 72.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 73.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 66.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 62.57,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Derived approximately from the supplied Vals category breakdown; marked with a tilde in the source sheet rather than treated as a directly reported headline score."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 60.07,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 52.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 56.98,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-excel-modeling-benchmark-vals-overall-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "benchmarkVersion": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 71.14,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/google_gemini-3.7-flash",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Vals AI Excel Modeling overall evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Excel Modeling overall evaluation.",
      "notes": "Derived as the exact mean of the seven published Vals workflow scores: 59, 66, 83, 58, 75, 86, and 71."
    },
    {
      "id": "observation-taxeval-v2-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "taxeval-v2-vals",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 76.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-taxeval-v2-vals-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "taxeval-v2-vals",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 74.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-taxeval-v2-vals-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "taxeval-v2-vals",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 75.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-taxeval-v2-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "taxeval-v2-vals",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 75.72,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-taxeval-v2-vals-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 71.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-taxeval-v2-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 75.55,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-taxeval-v2-vals-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "taxeval-v2-vals",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 73.06,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-taxeval-v2-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "benchmarkVersion": "TaxEval v2 · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 80.38,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI max-compute evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI max-compute evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medcode-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "medcode-vals",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 56.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medcode-vals-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "medcode-vals",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 44,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medcode-vals-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "medcode-vals",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 63.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medcode-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "medcode-vals",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 48.88,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medcode-vals-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "medcode-vals",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 44.71,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medcode-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "medcode-vals",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 40.67,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medcode-vals-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "medcode-vals",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 42.47,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medcode-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "medcode-vals",
      "benchmarkVersion": "MedCode · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 49.35,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medscribe-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "medscribe-vals",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 88.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medscribe-vals-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "medscribe-vals",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 85.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medscribe-vals-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "medscribe-vals",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 91,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medscribe-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "medscribe-vals",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 87.96,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medscribe-vals-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "medscribe-vals",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 86.53,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medscribe-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "medscribe-vals",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 84.95,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medscribe-vals-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "medscribe-vals",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 80.17,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-medscribe-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "medscribe-vals",
      "benchmarkVersion": "MedScribe · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 90.06,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI health evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI health evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mmlu-pro-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "mmlu-pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 91.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI five-shot MMLU-Pro harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI five-shot MMLU-Pro harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mmlu-pro-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "mmlu-pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 89.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI five-shot MMLU-Pro harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI five-shot MMLU-Pro harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mmlu-pro-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "mmlu-pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 91.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI five-shot MMLU-Pro harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI five-shot MMLU-Pro harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mmlu-pro-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "mmlu-pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 87.97,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI five-shot MMLU-Pro harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI five-shot MMLU-Pro harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mmlu-pro-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "mmlu-pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 89.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI five-shot MMLU-Pro harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI five-shot MMLU-Pro harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mmlu-pro-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "mmlu-pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 88.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI five-shot MMLU-Pro harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI five-shot MMLU-Pro harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mmlu-pro-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "mmlu-pro",
      "benchmarkVersion": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 88.28,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI five-shot MMLU-Pro harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI five-shot MMLU-Pro harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-index-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 75.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-index-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 73.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-index-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 74.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-index-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 74.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-index-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 71.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-index-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 66.12,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals-index",
      "sourceType": "third_party",
      "sourceName": "Memeburn",
      "sourceUrl": "https://memeburn.com/kimi-k3-vs-qwen-3-8/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Historical value: 66.12% is the early Qwen3.8 Max Vals Index listing, corroborated by Vals AI at 66.1% rounded. It is not Vals Index v2."
    },
    {
      "id": "observation-vals-index-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 66.25,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-index-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 71.88,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Daggered value: 71.88% matches the older Vals Index snapshot used by the existing peer values. The current Muse Spark 1.2 index is a different methodology/version and is not substituted."
    },
    {
      "id": "observation-vals-index-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "vals-index",
      "benchmarkVersion": "Vals Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 59.31,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/google_gemini-3.7-flash",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Vals AI index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI index methodology.",
      "notes": "Asterisked value: 59.31% is Vals Index v2, released after the older index snapshot represented by the other peer values; it is not apples-to-apples with the existing row."
    },
    {
      "id": "observation-vals-multimodal-index-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "vals-multimodal-index",
      "benchmarkVersion": "Vals Multimodal Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 74.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI multimodal index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI multimodal index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-multimodal-index-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "vals-multimodal-index",
      "benchmarkVersion": "Vals Multimodal Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 72.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI multimodal index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI multimodal index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-multimodal-index-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "vals-multimodal-index",
      "benchmarkVersion": "Vals Multimodal Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 73.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI multimodal index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI multimodal index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-multimodal-index-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "vals-multimodal-index",
      "benchmarkVersion": "Vals Multimodal Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 73.42,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI multimodal index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI multimodal index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-multimodal-index-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "benchmarkVersion": "Vals Multimodal Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 65.39,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI multimodal index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI multimodal index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-vals-multimodal-index-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "benchmarkVersion": "Vals Multimodal Index snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 69.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI multimodal index methodology",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI multimodal index methodology.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 49.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 48.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 55.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 44.23,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 48.08,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 47.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 40.87,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 43.75,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legal-research-bench-vals-overall-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "benchmarkVersion": "Vals Legal Research Bench · overall snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 34.62,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/google_gemini-3.7-flash",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "scaffold": "Vals AI Legal Research overall harness",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI Legal Research overall harness.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legalbench-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "legalbench-vals",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 88.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI LegalBench evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI LegalBench evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legalbench-vals-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "legalbench-vals",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 87,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI LegalBench evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI LegalBench evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legalbench-vals-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "legalbench-vals",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 87,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI LegalBench evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI LegalBench evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legalbench-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "legalbench-vals",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 86.02,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI LegalBench evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI LegalBench evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legalbench-vals-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "legalbench-vals",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 86.31,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI LegalBench evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI LegalBench evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legalbench-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "legalbench-vals",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 83.61,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI LegalBench evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI LegalBench evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legalbench-vals-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "legalbench-vals",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 82.36,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI LegalBench evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI LegalBench evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-legalbench-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "legalbench-vals",
      "benchmarkVersion": "LegalBench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 85.26,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI LegalBench evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI LegalBench evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-public-benefits-bench-vals-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "public-benefits-bench-vals",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 70.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI public-benefits evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI public-benefits evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-public-benefits-bench-vals-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "public-benefits-bench-vals",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 66.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI public-benefits evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI public-benefits evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-public-benefits-bench-vals-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "public-benefits-bench-vals",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 76.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI public-benefits evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI public-benefits evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-public-benefits-bench-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "public-benefits-bench-vals",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 68.27,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI public-benefits evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI public-benefits evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-public-benefits-bench-vals-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 66.85,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI public-benefits evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI public-benefits evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-public-benefits-bench-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 67.12,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI public-benefits evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI public-benefits evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-public-benefits-bench-vals-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "public-benefits-bench-vals",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 62.92,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Vals AI public-benefits evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI public-benefits evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-public-benefits-bench-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "benchmarkVersion": "Public Benefits Bench · Vals snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 68.47,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Vals AI public-benefits evaluation",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Vals AI public-benefits evaluation.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-aiiq-composite-iq-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "aiiq-composite-iq",
      "benchmarkVersion": "AIIQ Composite IQ snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 134,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "AIIQ composite snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: AIIQ composite snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-aiiq-composite-iq-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "aiiq-composite-iq",
      "benchmarkVersion": "AIIQ Composite IQ snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 136,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "AIIQ composite snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: AIIQ composite snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-aiiq-composite-iq-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "aiiq-composite-iq",
      "benchmarkVersion": "AIIQ Composite IQ snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 134,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "AIIQ composite snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: AIIQ composite snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-aiiq-composite-iq-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "aiiq-composite-iq",
      "benchmarkVersion": "AIIQ Composite IQ snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 122,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "AIIQ composite snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: AIIQ composite snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-design-arena-elo-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "design-arena-elo",
      "benchmarkVersion": "Design Arena rating snapshot · 2026-08-11",
      "status": "available",
      "rawScore": 1400,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Design Arena 2026-08-11 rating snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Design Arena 2026-08-11 rating snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-design-arena-elo-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "design-arena-elo",
      "benchmarkVersion": "Design Arena rating snapshot · 2026-08-11",
      "status": "available",
      "rawScore": 1379,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Design Arena 2026-08-11 rating snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Design Arena 2026-08-11 rating snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-design-arena-elo-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "design-arena-elo",
      "benchmarkVersion": "Design Arena rating snapshot · 2026-08-11",
      "status": "available",
      "rawScore": 1407,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Design Arena 2026-08-11 rating snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Design Arena 2026-08-11 rating snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-design-arena-elo-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "design-arena-elo",
      "benchmarkVersion": "Design Arena rating snapshot · 2026-08-11",
      "status": "available",
      "rawScore": 1453,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Design Arena 2026-08-11 rating snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Design Arena 2026-08-11 rating snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-design-arena-elo-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "design-arena-elo",
      "benchmarkVersion": "Design Arena rating snapshot · 2026-08-11",
      "status": "available",
      "rawScore": 1388,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Design Arena 2026-08-11 rating snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Design Arena 2026-08-11 rating snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-design-arena-elo-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "design-arena-elo",
      "benchmarkVersion": "Design Arena rating snapshot · 2026-08-11",
      "status": "available",
      "rawScore": 1373,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "scaffold": "Design Arena 2026-08-11 rating snapshot",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Design Arena 2026-08-11 rating snapshot.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-frontier-bench-v0-1-anthropic-h2h-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "benchmarkVersion": "Frontier-Bench v0.1 · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 33.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "benchmarkVersion": "Frontier-Bench v0.1 · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 34.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-frontier-bench-v0-1-anthropic-h2h-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "benchmarkVersion": "Frontier-Bench v0.1 · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 43.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-osworld-2-anthropic-h2h-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "benchmarkVersion": "OSWorld 2.0 · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 66.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-osworld-2-anthropic-h2h-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "benchmarkVersion": "OSWorld 2.0 · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 62.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-osworld-2-anthropic-h2h-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "benchmarkVersion": "OSWorld 2.0 · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 70.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-osworld-2-anthropic-h2h-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "benchmarkVersion": "OSWorld 2.0 · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 58.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "kimi-k3",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-osworld-2-anthropic-h2h-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "benchmarkVersion": "OSWorld 2.0 · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 46.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-osworld",
      "sourceType": "third_party",
      "sourceName": "Forward Future",
      "sourceUrl": "https://signals.forwardfuture.com/qwen3-8-benchmarks/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Qwen reports 19.4% binary and 46.7% partial on OSWorld 2.0; the partial score is used to match the existing Fable/Sol comparison metric."
    },
    {
      "id": "observation-osworld-verified-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "osworld-verified",
      "benchmarkVersion": "OSWorld-Verified · cross-source snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 85,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark family and success metric, but agent scaffolds and run snapshots differ.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-osworld-verified-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "osworld-verified",
      "benchmarkVersion": "OSWorld-Verified · cross-source snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 83.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark family and success metric, but agent scaffolds and run snapshots differ.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-osworld-verified-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "osworld-verified",
      "benchmarkVersion": "OSWorld-Verified · cross-source snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 90.69,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-ouroboros-osworld-opus-5",
      "sourceType": "third_party",
      "sourceName": "Ouroboros",
      "sourceUrl": "https://huggingface.co/datasets/razzant/ouroboros-osworld-verified-opus5",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "Claude Opus 5 adaptive thinking",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark family and success metric, but agent scaffolds and run snapshots differ.",
      "notes": "Self-reported 90.69% Ouroboros v6.87.0 run over all 361 accepted tasks: screenshot-only, one rollout, 100 policy turns, official evaluator, and max acting effort."
    },
    {
      "id": "observation-osworld-verified-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "osworld-verified",
      "benchmarkVersion": "OSWorld-Verified · cross-source snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 84.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark family and success metric, but agent scaffolds and run snapshots differ.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-osworld-verified-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "osworld-verified",
      "benchmarkVersion": "OSWorld-Verified · cross-source snapshot · 2026-08-16",
      "status": "available",
      "rawScore": 86.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, OpenLM.ai, and Artificial Analysis",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same benchmark family and success metric, but agent scaffolds and run snapshots differ.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-automationbench-v1-0-6-public-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "status": "available",
      "rawScore": 46.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same public benchmark version, with results assembled from the official leaderboard and a provider comparison table.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-automationbench-v1-0-6-public-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "status": "available",
      "rawScore": 45.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same public benchmark version, with results assembled from the official leaderboard and a provider comparison table.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-automationbench-v1-0-6-public-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "status": "available",
      "rawScore": 50.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-automationbench-public",
      "sourceType": "benchmark_org",
      "sourceName": "Zapier",
      "sourceUrl": "https://github.com/zapier/AutomationBench",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same public benchmark version, with results assembled from the official leaderboard and a provider comparison table.",
      "notes": "Official public v1.0.6 score is 50.3%; the 26.0% figure belongs to the distinct private held-out evaluation."
    },
    {
      "id": "observation-automationbench-v1-0-6-public-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "status": "available",
      "rawScore": 46.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "kimi-k3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same public benchmark version, with results assembled from the official leaderboard and a provider comparison table.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-automationbench-v1-0-6-public-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "status": "available",
      "rawScore": 39.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same public benchmark version, with results assembled from the official leaderboard and a provider comparison table.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-automationbench-v1-0-6-public-glm-5-3-max",
      "fixture": false,
      "modelId": "glm-5-3-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "status": "available",
      "rawScore": 48.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "glm-5.3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same public benchmark version, with results assembled from the official leaderboard and a provider comparison table.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-automationbench-v1-0-6-public-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "status": "available",
      "rawScore": 43.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Same public benchmark version, with results assembled from the official leaderboard and a provider comparison table.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-automationbench-v1-0-6-public-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "benchmarkVersion": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16",
      "status": "available",
      "rawScore": 30.44,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Same public benchmark version, with results assembled from the official leaderboard and a provider comparison table.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-automationbench-anthropic-h2h-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "benchmarkVersion": "AutomationBench · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 17.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-automationbench-anthropic-h2h-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "benchmarkVersion": "AutomationBench · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 18.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-automationbench-anthropic-h2h-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "benchmarkVersion": "AutomationBench · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 26,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published head-to-head setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published head-to-head setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-harvey-legal-agent-held-out-h2h-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "benchmarkVersion": "Legal Agent Benchmark held-out · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 13.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published held-out legal-agent setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published held-out legal-agent setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-harvey-legal-agent-held-out-h2h-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "benchmarkVersion": "Legal Agent Benchmark held-out · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 2.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published held-out legal-agent setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published held-out legal-agent setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-harvey-legal-agent-held-out-h2h-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "benchmarkVersion": "Legal Agent Benchmark held-out · Anthropic Opus 5 head-to-head",
      "status": "available",
      "rawScore": 11.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "sourceType": "third_party",
      "sourceName": "Tosea.ai",
      "sourceUrl": "https://tosea.ai/blog/claude-opus-5-complete-guide",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "published max configuration",
      "reasoningBudget": "max effort",
      "scaffold": "Anthropic Opus 5 published held-out legal-agent setup",
      "comparability": "comparable",
      "comparabilityNote": "Same published comparison family: Anthropic Opus 5 published held-out legal-agent setup.",
      "notes": "Value imported from the requested comparison sheet under the named evaluator and harness."
    },
    {
      "id": "observation-mcp-atlas-cross-source-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "benchmarkVersion": "MCP Atlas · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 84.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-mcp-atlas-cross-source-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "benchmarkVersion": "MCP Atlas · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 83.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-mcp-atlas-cross-source-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "benchmarkVersion": "MCP Atlas · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 85.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-mcp-atlas-cross-source-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "benchmarkVersion": "MCP Atlas · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 84.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-mcp-atlas-cross-source-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "benchmarkVersion": "MCP Atlas · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 73.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-v4-pro-card",
      "sourceType": "vendor",
      "sourceName": "DeepSeek",
      "sourceUrl": "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro maximum reasoning effort",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-mcp-atlas-cross-source-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "benchmarkVersion": "MCP Atlas · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 90.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-meta-muse-spark-1-2-methodology",
      "sourceType": "vendor",
      "sourceName": "Meta",
      "sourceUrl": "https://research.meta.ai/static/muse-spark-1-2-methodology",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-code-migration-cross-source-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "code-migration-cross-source",
      "benchmarkVersion": "Code Migration · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 55.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-code-migration-cross-source-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "code-migration-cross-source",
      "benchmarkVersion": "Code Migration · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 52.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-code-migration-cross-source-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "code-migration-cross-source",
      "benchmarkVersion": "Code Migration · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 57.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-code-migration-cross-source-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "code-migration-cross-source",
      "benchmarkVersion": "Code Migration · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 16.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-code-migration-cross-source-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "benchmarkVersion": "Code Migration · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 44.57,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchmarkList, BenchLM, Vals AI, and Artificial Analysis",
      "sourceUrl": "https://benchlm.ai/compare/grok-4-6-vs-inkling",
      "modelVersion": "grok-4.6",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-code-migration-cross-source-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "code-migration-cross-source",
      "benchmarkVersion": "Code Migration · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 41.54,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-code-migration-cross-source-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "benchmarkVersion": "Code Migration · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 29.95,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-skillsbench-cross-source-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "skillsbench-cross-source",
      "benchmarkVersion": "SkillsBench · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 70.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-skillsbench-cross-source-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "skillsbench-cross-source",
      "benchmarkVersion": "SkillsBench · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 73.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-skillsbench-cross-source-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "skillsbench-cross-source",
      "benchmarkVersion": "SkillsBench · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 60.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-skillsbench-cross-source-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "benchmarkVersion": "SkillsBench · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 70.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://benchlm.ai/benchmarks/valsswebench",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-skillsbench-cross-source-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "benchmarkVersion": "SkillsBench · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 53.04,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals AI / BenchLM",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-officeqa-pro-cross-source-claude-fable-5-max",
      "fixture": false,
      "modelId": "claude-fable-5-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "benchmarkVersion": "OfficeQA Pro · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 69.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-fable-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Source marks this 69.9% result with an asterisk and identifies max configuration with fallback."
    },
    {
      "id": "observation-officeqa-pro-cross-source-gpt-5-6-sol-max",
      "fixture": false,
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "benchmarkVersion": "OfficeQA Pro · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 63.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "gpt-5.6-sol",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Source marks this 63.2% result with an asterisk; the marker is retained as a provenance caveat."
    },
    {
      "id": "observation-officeqa-pro-cross-source-claude-opus-5-max",
      "fixture": false,
      "modelId": "claude-opus-5-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "benchmarkVersion": "OfficeQA Pro · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 66.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-fable-5",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/anthropic-claude-fable-5/",
      "modelVersion": "claude-opus-5",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-officeqa-pro-cross-source-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "benchmarkVersion": "OfficeQA Pro · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 63.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "independent",
      "sourceName": "BenchmarkList",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "Value imported from the requested comparison sheet and linked to the public comparison source; it remains display-only because full harness metadata is unavailable."
    },
    {
      "id": "observation-officeqa-pro-cross-source-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "benchmarkVersion": "OfficeQA Pro · cross-source snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 63.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-model-card",
      "sourceType": "vendor",
      "sourceName": "SpaceXAI (xAI)",
      "sourceUrl": "https://media.x.ai/v1/website/card-7f81d41b.pdf",
      "modelVersion": "grok-4.6",
      "reasoningMode": "comparison-table configuration; effort not independently pinned",
      "comparability": "limited",
      "comparabilityNote": "Public values are sufficiently similar for display but are not established as one controlled evaluator run.",
      "notes": "xAI’s Grok 4.6 model card reports 63.2% OfficeQA Pro accuracy at High effort; retained as cross-source evidence."
    },
    {
      "id": "observation-gdpval-aa-v2-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1682,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-briefcase-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "aa-briefcase",
      "benchmarkVersion": "AA-Briefcase public leaderboard snapshot",
      "status": "available",
      "rawScore": 1541,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-intelligence-index-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "aa-intelligence-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 60,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "Kimi K3 is retained against the existing benchmark definition and version; the source record is still preserved below.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-frontiermath-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "frontiermath",
      "benchmarkVersion": "FrontierMath v2",
      "status": "available",
      "rawScore": 39.02,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "Kimi K3 is retained against the existing benchmark definition and version; the source record is still preserved below.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-deepswe-1-1-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 69,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "Kimi K3 is retained against the existing benchmark definition and version; the source record is still preserved below.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-cursorbench-3-2-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 60.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-apex-agents-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "apex-agents",
      "benchmarkVersion": "APEX-Agents public leaderboard snapshot",
      "status": "available",
      "rawScore": 55.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-terminal-bench-3-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "terminal-bench-3",
      "benchmarkVersion": "Terminal-Bench v3.0",
      "status": "available",
      "rawScore": 17.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-explainx-terminal-bench-3",
      "sourceType": "third_party",
      "sourceName": "ExplainX transcription of Z.ai comparison",
      "sourceUrl": "https://explainx.ai/blog/glm-5-3-launch-cyber-defense-benchmarks-august-2026",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Asterisked value: same Terminal-Bench 3.0 benchmark, but the source uses a slightly different comparison snapshot from the existing Fable/Sol/Opus row."
    },
    {
      "id": "observation-agents-last-exam-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "agents-last-exam",
      "benchmarkVersion": "Agents' Last Exam · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 28.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-snorkel-agents-last-exam",
      "sourceType": "benchmark_org",
      "sourceName": "Snorkel AI Agents' Last Exam leaderboard",
      "sourceUrl": "https://snorkel.ai/leaderboard/agents-last-exam/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-harvey-lab-vals-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "harvey-lab-vals",
      "benchmarkVersion": "Vals Harvey LAB · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 10.83,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-frontierswe-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "frontierswe",
      "benchmarkVersion": "FrontierSWE · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 81.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-jobbench-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "jobbench",
      "benchmarkVersion": "JobBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 54.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-babyvision-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "babyvision",
      "benchmarkVersion": "BabyVision (with CI) · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 85.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-charxiv-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "charxiv",
      "benchmarkVersion": "CharXiv v1.0",
      "status": "available",
      "rawScore": 91.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-perceptionbench-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "perceptionbench",
      "benchmarkVersion": "PerceptionBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 58.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-apex-swe-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "apex-swe",
      "benchmarkVersion": "APEX-SWE public leaderboard snapshot",
      "status": "available",
      "rawScore": 48,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-browsecomp-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "browsecomp",
      "benchmarkVersion": "BrowseComp public release",
      "status": "available",
      "rawScore": 91.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-toolathlon-verified-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "toolathlon-verified",
      "benchmarkVersion": "Toolathlon-Verified · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 76.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-mmmu-pro-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "mmmu-pro",
      "benchmarkVersion": "Vals MMMU-Pro · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 81.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-humanitys-last-exam-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "humanitys-last-exam",
      "benchmarkVersion": "HLE public leaderboard snapshot · 2,500-question release",
      "status": "available",
      "rawScore": 44.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-tau2-bench-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "tau2-bench",
      "benchmarkVersion": "τ²-bench public leaderboard snapshot",
      "status": "available",
      "rawScore": 80.63,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-gpqa-diamond-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 93.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "Kimi K3 is retained against the existing benchmark definition and version; the source record is still preserved below.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-omniscience-index-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "aa-omniscience-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 18.42,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "Kimi K3 is retained against the existing benchmark definition and version; the source record is still preserved below.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-lcr-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "aa-lcr",
      "benchmarkVersion": "AA-LCR · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 74.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-arc-agi-2-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "arc-agi-2",
      "benchmarkVersion": "ARC-AGI-2 · public results snapshot",
      "status": "available",
      "rawScore": 60.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "Kimi K3 is retained against the existing benchmark definition and version; the source record is still preserved below.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-chatbot-arena-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "chatbot-arena",
      "benchmarkVersion": "Chatbot Arena public leaderboard snapshot",
      "status": "available",
      "rawScore": 1489,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-livebench-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "livebench",
      "benchmarkVersion": "LiveBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 79.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-livecodebench-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "livecodebench",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 87.19,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-creative-writing-v3-kimi-k3-max",
      "fixture": false,
      "modelId": "kimi-k3-max",
      "benchmarkId": "creative-writing-v3",
      "benchmarkVersion": "Creative Writing v3 · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 2340.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-benchmarklist-kimi-k3",
      "sourceType": "third_party",
      "sourceName": "BenchmarkList Kimi K3 model record",
      "sourceUrl": "https://benchmarklist.com/models/moonshotai-kimi-k3/",
      "modelVersion": "kimi-k3",
      "reasoningMode": "Kimi K3 published max reasoning configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "Kimi K3 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Kimi K3 source-matched benchmark sheet row."
    },
    {
      "id": "observation-apex-swe-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "apex-swe",
      "benchmarkVersion": "APEX-SWE public leaderboard snapshot",
      "status": "available",
      "rawScore": 56.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Grok 4.6 High official benchmark report",
      "sourceUrl": "https://x.ai/news/grok-4-6",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Grok 4.6 High is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Grok 4.6 High source-matched benchmark sheet row."
    },
    {
      "id": "observation-gpqa-diamond-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 94.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis GPQA Diamond leaderboard",
      "sourceUrl": "https://artificialanalysis.ai/evaluations/gpqa-diamond",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "comparable",
      "comparabilityNote": "Grok 4.6 High is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested Grok 4.6 High source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-omniscience-index-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 30.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Grok 4.6 High benchmark comparison",
      "sourceUrl": "https://docsbot.ai/models/compare/grok-4-6/kimi-k3",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "comparable",
      "comparabilityNote": "Grok 4.6 High is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested Grok 4.6 High source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-lcr-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "aa-lcr",
      "benchmarkVersion": "AA-LCR · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 75,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Grok 4.6 High benchmark comparison",
      "sourceUrl": "https://llmbase.ai/compare/grok-4-6%2Ckimi-k2-7-code/",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Grok 4.6 High is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Grok 4.6 High source-matched benchmark sheet row."
    },
    {
      "id": "observation-arc-agi-2-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "arc-agi-2",
      "benchmarkVersion": "ARC-AGI-2 · public results snapshot",
      "status": "available",
      "rawScore": 67.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "ARC Prize Grok 4.6 High result",
      "sourceUrl": "https://x.com/arcprize?lang=en",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "comparable",
      "comparabilityNote": "Grok 4.6 High is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested Grok 4.6 High source-matched benchmark sheet row."
    },
    {
      "id": "observation-chatbot-arena-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "chatbot-arena",
      "benchmarkVersion": "Chatbot Arena public leaderboard snapshot",
      "status": "available",
      "rawScore": 1464,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-arena",
      "sourceType": "benchmark_org",
      "sourceName": "Arena AI text leaderboard",
      "sourceUrl": "https://arena.ai/leaderboard/text",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Grok 4.6 High is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Preliminary/live LMArena score; it may move as additional votes accumulate."
    },
    {
      "id": "observation-livebench-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "livebench",
      "benchmarkVersion": "LiveBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 78,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "LiveBench leaderboard",
      "sourceUrl": "https://livebench.ai/",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Grok 4.6 High is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Grok 4.6 High source-matched benchmark sheet row."
    },
    {
      "id": "observation-livecodebench-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "livecodebench",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 88.22,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "BenchLM Vals LiveCodeBench mirror",
      "sourceUrl": "https://benchlm.ai/benchmarks/valslivecodebench",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Grok 4.6 High is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Grok 4.6 High source-matched benchmark sheet row."
    },
    {
      "id": "observation-humanitys-last-exam-grok-4-6-xhigh",
      "fixture": false,
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "benchmarkVersion": "HLE public leaderboard snapshot · 2,500-question release",
      "status": "available",
      "rawScore": 42.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Grok 4.6 High HLE evaluation",
      "sourceUrl": "https://www.orcarouter.ai/models/grok/grok-4.6",
      "modelVersion": "grok-4.6",
      "reasoningMode": "Grok 4.6 High published configuration",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Grok 4.6 High is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Grok 4.6 High source-matched benchmark sheet row."
    },
    {
      "id": "observation-gdpval-aa-v2-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1737,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-briefcase-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "aa-briefcase",
      "benchmarkVersion": "AA-Briefcase public leaderboard snapshot",
      "status": "available",
      "rawScore": 1420,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-intelligence-index-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 58,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "comparable",
      "comparabilityNote": "Qwen3.8 Max is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-frontiermath-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "frontiermath",
      "benchmarkVersion": "FrontierMath v2",
      "status": "available",
      "rawScore": 46.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "uncertainty": {
        "ciLower": 38.4,
        "ciUpper": 54.2,
        "confidenceLevel": 0.95
      },
      "comparability": "comparable",
      "comparabilityNote": "Qwen3.8 Max is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Qwen3.8 Max result reported at xhigh effort as 46.3% ±7.9 percentage points; uncertainty is retained on the observation."
    },
    {
      "id": "observation-deepswe-1-1-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 56.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "comparable",
      "comparabilityNote": "Qwen3.8 Max is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-agents-last-exam-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "agents-last-exam",
      "benchmarkVersion": "Agents' Last Exam · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 27,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-snorkel-agents-last-exam",
      "sourceType": "benchmark_org",
      "sourceName": "Snorkel AI Agents' Last Exam leaderboard",
      "sourceUrl": "https://snorkel.ai/leaderboard/agents-last-exam/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-harvey-lab-vals-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "benchmarkVersion": "Vals Harvey LAB · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 10.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-frontierswe-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "frontierswe",
      "benchmarkVersion": "FrontierSWE · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 73.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-jobbench-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "jobbench",
      "benchmarkVersion": "JobBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 53.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-babyvision-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "babyvision",
      "benchmarkVersion": "BabyVision (with CI) · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 91.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Qwen reports the with-confidence-interval result; its companion no-CI value is 82.0%, while this row uses 91.3%."
    },
    {
      "id": "observation-charxiv-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "charxiv",
      "benchmarkVersion": "CharXiv v1.0",
      "status": "available",
      "rawScore": 93.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Qwen reports the with-CI / RQ result; its companion no-CI value is 88.4%, while this row uses 93.5%."
    },
    {
      "id": "observation-perceptionbench-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "perceptionbench",
      "benchmarkVersion": "PerceptionBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 63.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-toolathlon-verified-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "toolathlon-verified",
      "benchmarkVersion": "Toolathlon-Verified · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 72.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-mmmu-pro-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "mmmu-pro",
      "benchmarkVersion": "Vals MMMU-Pro · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 82.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-humanitys-last-exam-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "benchmarkVersion": "HLE public leaderboard snapshot · 2,500-question release",
      "status": "available",
      "rawScore": 43.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-gpqa-diamond-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 92.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "comparable",
      "comparabilityNote": "Qwen3.8 Max is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-lcr-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "aa-lcr",
      "benchmarkVersion": "AA-LCR · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 74.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-chatbot-arena-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "chatbot-arena",
      "benchmarkVersion": "Chatbot Arena public leaderboard snapshot",
      "status": "available",
      "rawScore": 1491,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Raw LMArena text Elo is approximately 1490.70 and is rounded to 1491 in the sheet."
    },
    {
      "id": "observation-livebench-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "livebench",
      "benchmarkVersion": "LiveBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 78.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-livebench",
      "sourceType": "benchmark_org",
      "sourceName": "LiveBench leaderboard",
      "sourceUrl": "https://livebench.ai/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "LiveBench-2026-06-25 live overall score; the value may change after a leaderboard refresh."
    },
    {
      "id": "observation-livecodebench-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "livecodebench",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 87.85,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max benchmark comparison",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-swe-bench-pro-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "swe-bench-pro",
      "benchmarkVersion": "Pro · public leaderboard snapshot",
      "status": "available",
      "rawScore": 67.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max model-card benchmark table",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-paperbench-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "paperbench",
      "benchmarkVersion": "PaperBench (Replication Score) · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 93,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max model-card benchmark table",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-qwenreactbench-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "qwenreactbench",
      "benchmarkVersion": "QwenReactBench (Elo) · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 1724,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max model-card benchmark table",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-coworkbench-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "coworkbench",
      "benchmarkVersion": "CoWorkBench · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 74.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "Qwen3.8 Max model-card benchmark table",
      "sourceUrl": "https://benchmarklist.com/models/qwen-qwen3.8-max/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-erqa-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "erqa",
      "benchmarkVersion": "ERQA · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 77.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "OpenLM.ai Qwen3.8 benchmark comparison",
      "sourceUrl": "https://openlm.ai/qwen3.8/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-lvbench-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "lvbench",
      "benchmarkVersion": "LVBench (with Memory) · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 85.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "OpenLM.ai Qwen3.8 benchmark comparison",
      "sourceUrl": "https://openlm.ai/qwen3.8/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-vision2web-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "vision2web",
      "benchmarkVersion": "Vision2Web (Avg. Frontend/Webpage/etc.) · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 69,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "OpenLM.ai Qwen3.8 benchmark comparison",
      "sourceUrl": "https://openlm.ai/qwen3.8/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-mobileworld-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "mobileworld",
      "benchmarkVersion": "MobileWorld · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 77.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "OpenLM.ai Qwen3.8 benchmark comparison",
      "sourceUrl": "https://openlm.ai/qwen3.8/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-webarena-verified-qwen-3-8-max-xhigh",
      "fixture": false,
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "webarena-verified",
      "benchmarkVersion": "WebArena-Verified · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 66.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "sourceType": "independent",
      "sourceName": "OpenLM.ai Qwen3.8 benchmark comparison",
      "sourceUrl": "https://openlm.ai/qwen3.8/",
      "modelVersion": "qwen3.8-max",
      "reasoningMode": "Qwen3.8 Max published configuration",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Qwen3.8 Max is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Qwen3.8 Max source-matched benchmark sheet row."
    },
    {
      "id": "observation-gdpval-aa-v2-glm-5-3-max",
      "fixture": false,
      "modelId": "glm-5-3-max",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1769,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai GLM-5.3 launch benchmark report",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "glm-5.3",
      "reasoningMode": "GLM-5.3 published provider configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "GLM-5.3 is retained as a provider-published row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested GLM-5.3 provider benchmark row."
    },
    {
      "id": "observation-deepswe-1-1-glm-5-3-max",
      "fixture": false,
      "modelId": "glm-5-3-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 66.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai GLM-5.3 launch benchmark report",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "glm-5.3",
      "reasoningMode": "GLM-5.3 published provider configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "GLM-5.3 is retained against the existing benchmark definition and version; the provider source record is preserved below.",
      "notes": "Value imported from the requested GLM-5.3 provider benchmark row."
    },
    {
      "id": "observation-terminal-bench-3-glm-5-3-max",
      "fixture": false,
      "modelId": "glm-5-3-max",
      "benchmarkId": "terminal-bench-3",
      "benchmarkVersion": "Terminal-Bench v3.0",
      "status": "available",
      "rawScore": 28.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai GLM-5.3 launch benchmark report",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "glm-5.3",
      "reasoningMode": "GLM-5.3 published provider configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "GLM-5.3 is retained as a provider-published row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested GLM-5.3 provider benchmark row."
    },
    {
      "id": "observation-agents-last-exam-glm-5-3-max",
      "fixture": false,
      "modelId": "glm-5-3-max",
      "benchmarkId": "agents-last-exam",
      "benchmarkVersion": "Agents' Last Exam · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 28.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai GLM-5.3 launch benchmark report",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "glm-5.3",
      "reasoningMode": "GLM-5.3 published provider configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "GLM-5.3 is retained as a provider-published row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Asterisked value: Z.ai labels this as Agents’ Last Exam (CLI); the 28.5% value is retained as the pass rate."
    },
    {
      "id": "observation-frontierswe-glm-5-3-max",
      "fixture": false,
      "modelId": "glm-5-3-max",
      "benchmarkId": "frontierswe",
      "benchmarkVersion": "FrontierSWE · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 78.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai GLM-5.3 launch benchmark report",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "glm-5.3",
      "reasoningMode": "GLM-5.3 published provider configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "GLM-5.3 is retained as a provider-published row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested GLM-5.3 provider benchmark row."
    },
    {
      "id": "observation-toolathlon-verified-glm-5-3-max",
      "fixture": false,
      "modelId": "glm-5-3-max",
      "benchmarkId": "toolathlon-verified",
      "benchmarkVersion": "Toolathlon-Verified · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 73,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-zai-glm-5-3",
      "sourceType": "vendor",
      "sourceName": "Z.ai GLM-5.3 launch benchmark report",
      "sourceUrl": "https://z.ai/blog/glm-5.3",
      "modelVersion": "glm-5.3",
      "reasoningMode": "GLM-5.3 published provider configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "GLM-5.3 is retained as a provider-published row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested GLM-5.3 provider benchmark row."
    },
    {
      "id": "observation-gdpval-aa-v2-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1590,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis DeepSeek V4 Pro 0813 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/deepseek-v4-pro",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-intelligence-index-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "aa-intelligence-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 53,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-v4-pro-0813",
      "sourceType": "vendor",
      "sourceName": "DeepSeek V4 Pro 0813 benchmark report",
      "sourceUrl": "https://api-docs.deepseek.com/news/news260813/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-deepswe-1-1-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 62.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-v4-pro-0813",
      "sourceType": "vendor",
      "sourceName": "DeepSeek V4 Pro 0813 benchmark report",
      "sourceUrl": "https://api-docs.deepseek.com/news/news260813/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-harvey-lab-vals-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "harvey-lab-vals",
      "benchmarkVersion": "Vals Harvey LAB · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 7.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals DeepSeek V4 Pro 0813 leaderboard",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-swe-bench-pro-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "swe-bench-pro",
      "benchmarkVersion": "Pro · public leaderboard snapshot",
      "status": "available",
      "rawScore": 55.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-v4-pro-0813",
      "sourceType": "vendor",
      "sourceName": "DeepSeek V4 Pro 0813 benchmark report",
      "sourceUrl": "https://api-docs.deepseek.com/news/news260813/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-agents-last-exam-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "agents-last-exam",
      "benchmarkVersion": "Agents' Last Exam · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 25.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-v4-pro-0813",
      "sourceType": "vendor",
      "sourceName": "DeepSeek V4 Pro 0813 benchmark report",
      "sourceUrl": "https://api-docs.deepseek.com/news/news260813/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-browsecomp-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "browsecomp",
      "benchmarkVersion": "BrowseComp public release",
      "status": "available",
      "rawScore": 83.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-v4-pro-0813",
      "sourceType": "vendor",
      "sourceName": "DeepSeek V4 Pro 0813 benchmark report",
      "sourceUrl": "https://api-docs.deepseek.com/news/news260813/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Published DeepSeek V4 Pro score; this is not a separately identified 0813 rerun."
    },
    {
      "id": "observation-toolathlon-verified-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "toolathlon-verified",
      "benchmarkVersion": "Toolathlon-Verified · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 74.1,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-v4-pro-0813",
      "sourceType": "vendor",
      "sourceName": "DeepSeek V4 Pro 0813 benchmark report",
      "sourceUrl": "https://api-docs.deepseek.com/news/news260813/",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-humanitys-last-exam-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "humanitys-last-exam",
      "benchmarkVersion": "HLE public leaderboard snapshot · 2,500-question release",
      "status": "available",
      "rawScore": 41,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis DeepSeek V4 Pro 0813 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/deepseek-v4-pro",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-gpqa-diamond-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 92.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis DeepSeek V4 Pro 0813 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/deepseek-v4-pro",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "comparable",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained against the existing benchmark definition and version; the source record is preserved below.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-lcr-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "aa-lcr",
      "benchmarkVersion": "AA-LCR · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 75.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis DeepSeek V4 Pro 0813 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/deepseek-v4-pro",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-chatbot-arena-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "chatbot-arena",
      "benchmarkVersion": "Chatbot Arena public leaderboard snapshot",
      "status": "available",
      "rawScore": 1465,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-arena",
      "sourceType": "benchmark_org",
      "sourceName": "Arena AI text leaderboard",
      "sourceUrl": "https://arena.ai/leaderboard/text",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "AutoEval text Elo for deepseek-v4-pro-max-20260813; uncertainty is approximately ±10 Elo."
    },
    {
      "id": "observation-livecodebench-deepseek-v4-pro-max",
      "fixture": false,
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "livecodebench",
      "benchmarkVersion": "Vals LiveCodeBench · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 87.53,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-deepseek-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals DeepSeek V4 Pro 0813 leaderboard",
      "sourceUrl": "https://www.vals.ai/models/deepseek_deepseek-v4-pro-0813",
      "modelVersion": "deepseek-v4-pro",
      "reasoningMode": "DeepSeek V4 Pro 0813 published configuration",
      "reasoningBudget": "max effort",
      "comparability": "limited",
      "comparabilityNote": "DeepSeek V4 Pro 0813 is retained as a source-matched row, but the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested DeepSeek V4 Pro 0813 source-matched benchmark sheet row."
    },
    {
      "id": "observation-gdpval-aa-v2-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1628,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Muse Spark 1.2 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Muse Spark 1.2 is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-briefcase-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "aa-briefcase",
      "benchmarkVersion": "AA-Briefcase public leaderboard snapshot",
      "status": "available",
      "rawScore": 1358,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Muse Spark 1.2 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Muse Spark 1.2 is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-intelligence-index-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 57,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Muse Spark 1.2 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "comparable",
      "comparabilityNote": "Muse Spark 1.2 is retained against the named evaluator and benchmark definition.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-deepswe-1-1-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 59.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-code",
      "sourceType": "vendor",
      "sourceName": "Muse Code benchmark report",
      "sourceUrl": "https://benchlm.ai/compare/muse-spark-1-1-vs-muse-spark-1-2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "comparable",
      "comparabilityNote": "Muse Spark 1.2 is retained against the named evaluator and benchmark definition.",
      "notes": "Asterisked value: Muse Spark 1.2 running with Muse Code; this is a model-plus-agent result rather than a bare-model comparison."
    },
    {
      "id": "observation-harvey-lab-vals-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "benchmarkVersion": "Vals Harvey LAB · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 25.42,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals Muse Spark 1.2 leaderboard",
      "sourceUrl": "https://www.vals.ai/models/meta_muse_spark_1_2",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Muse Spark 1.2 is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-humanitys-last-exam-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "benchmarkVersion": "HLE public leaderboard snapshot · 2,500-question release",
      "status": "available",
      "rawScore": 45.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Muse Spark 1.2 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Muse Spark 1.2 is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-gpqa-diamond-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 90.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Muse Spark 1.2 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "comparable",
      "comparabilityNote": "Muse Spark 1.2 is retained against the named evaluator and benchmark definition.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-omniscience-index-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 27.2,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Muse Spark 1.2 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "comparable",
      "comparabilityNote": "Muse Spark 1.2 is retained against the named evaluator and benchmark definition.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-lcr-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "aa-lcr",
      "benchmarkVersion": "AA-LCR · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 83.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Muse Spark 1.2 evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/muse-spark-1-2-vs-mimo-v2-5-0424",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Muse Spark 1.2 is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-chatbot-arena-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "chatbot-arena",
      "benchmarkVersion": "Chatbot Arena public leaderboard snapshot",
      "status": "available",
      "rawScore": 1499,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-arena",
      "sourceType": "benchmark_org",
      "sourceName": "BenchmarkList Muse Spark 1.2 profile",
      "sourceUrl": "https://benchmarklist.com/models/meta-muse-spark-1.2/",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Muse Spark 1.2 is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Rounded from the reported 1498.64 text Elo."
    },
    {
      "id": "observation-livebench-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "livebench",
      "benchmarkVersion": "LiveBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 78,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-livebench",
      "sourceType": "benchmark_org",
      "sourceName": "LiveBench Muse Spark 1.2 leaderboard",
      "sourceUrl": "https://livebench.ai/",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Muse Spark 1.2 is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Muse Spark 1.2 source-matched benchmark sheet row."
    },
    {
      "id": "observation-toolathlon-verified-muse-spark-1-2-xhigh",
      "fixture": false,
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "toolathlon-verified",
      "benchmarkVersion": "Toolathlon-Verified · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 75.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-muse-spark-1-2-toolathlon",
      "sourceType": "benchmark_org",
      "sourceName": "Toolathlon-Verified leaderboard",
      "sourceUrl": "https://toolathlon.xyz/docs/leaderboard",
      "modelVersion": "muse-spark-1.2",
      "reasoningMode": "Muse Spark 1.2 extra-high effort",
      "reasoningBudget": "xhigh effort",
      "comparability": "limited",
      "comparabilityNote": "Muse Spark 1.2 is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Independently evaluated Pass@1 result at xhigh effort with Toolathlon’s default agent, dated 2026-08-05."
    },
    {
      "id": "observation-gdpval-aa-v2-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "gdpval-aa-v2",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GDPval-AA v2",
      "status": "available",
      "rawScore": 1525,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Gemini 3.7 Flash evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-briefcase-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "aa-briefcase",
      "benchmarkVersion": "AA-Briefcase public leaderboard snapshot",
      "status": "available",
      "rawScore": 1132,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Gemini 3.7 Flash evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-intelligence-index-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "aa-intelligence-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 56,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Gemini 3.7 Flash evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "comparable",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained against the named evaluator and benchmark definition.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-cursorbench-3-2-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "cursorbench-3-2",
      "benchmarkVersion": "CursorBench 3.2",
      "status": "available",
      "rawScore": 61.6,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-cursorbench-32",
      "sourceType": "benchmark_org",
      "sourceName": "CursorBench 3.2 leaderboard",
      "sourceUrl": "https://cursor.com/cursorbench",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "comparable",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained against the named evaluator and benchmark definition.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-deepswe-1-1-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "deepswe-1-1",
      "benchmarkVersion": "DeepSWE v1.1",
      "status": "available",
      "rawScore": 65.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-google",
      "sourceType": "vendor",
      "sourceName": "Google DeepMind Gemini 3.7 Flash model card",
      "sourceUrl": "https://deepmind.google/models/gemini/flash/",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "comparable",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained against the named evaluator and benchmark definition.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-terminal-bench-3-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "terminal-bench-3",
      "benchmarkVersion": "Terminal-Bench v3.0",
      "status": "available",
      "rawScore": 14.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-google",
      "sourceType": "vendor",
      "sourceName": "Google DeepMind Gemini 3.7 Flash model card",
      "sourceUrl": "https://deepmind.google/models/gemini/flash/",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-harvey-lab-vals-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "harvey-lab-vals",
      "benchmarkVersion": "Vals Harvey LAB · public leaderboard snapshot · 2026-08-15",
      "status": "available",
      "rawScore": 8.75,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-vals",
      "sourceType": "benchmark_org",
      "sourceName": "Vals Gemini 3.7 Flash leaderboard",
      "sourceUrl": "https://www.vals.ai/models/google_gemini-3.7-flash",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-agents-last-exam-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "agents-last-exam",
      "benchmarkVersion": "Agents' Last Exam · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 26.3,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-google",
      "sourceType": "vendor",
      "sourceName": "Google DeepMind Gemini 3.7 Flash model card",
      "sourceUrl": "https://deepmind.google/models/gemini/flash/",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-charxiv-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "charxiv",
      "benchmarkVersion": "CharXiv v1.0",
      "status": "available",
      "rawScore": 88.7,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-google",
      "sourceType": "vendor",
      "sourceName": "Google DeepMind Gemini 3.7 Flash model card",
      "sourceUrl": "https://deepmind.google/models/gemini/flash/",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Asterisked value: Google reports Gemini 3.7 Flash with tools; the protocol is not identical to the row’s generic with-CI/RQ label."
    },
    {
      "id": "observation-lvbench-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "lvbench",
      "benchmarkVersion": "LVBench (with Memory) · Qwen3.8 Max source snapshot",
      "status": "available",
      "rawScore": 85.4,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-google",
      "sourceType": "vendor",
      "sourceName": "Google DeepMind Gemini 3.7 Flash model card",
      "sourceUrl": "https://deepmind.google/models/gemini/flash/",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Asterisked value: direct Gemini 3.7 Flash LVBench result; the source does not establish the row’s with-Memory protocol."
    },
    {
      "id": "observation-humanitys-last-exam-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "humanitys-last-exam",
      "benchmarkVersion": "HLE public leaderboard snapshot · 2,500-question release",
      "status": "available",
      "rawScore": 47.9,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Gemini 3.7 Flash evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-gpqa-diamond-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "gpqa-diamond",
      "benchmarkVersion": "AA Intelligence Index v4.1.1 · GPQA Diamond",
      "status": "available",
      "rawScore": 94.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Gemini 3.7 Flash evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "comparable",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained against the named evaluator and benchmark definition.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-aa-omniscience-index-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "aa-omniscience-index",
      "benchmarkVersion": "AA Intelligence Index v4.1.1",
      "status": "available",
      "rawScore": 26.5,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Gemini 3.7 Flash evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "comparable",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained against the named evaluator and benchmark definition.",
      "notes": "Asterisked value: dynamic Artificial Analysis leaderboard snapshot."
    },
    {
      "id": "observation-aa-lcr-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "aa-lcr",
      "benchmarkVersion": "AA-LCR · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 80,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-aa",
      "sourceType": "independent",
      "sourceName": "Artificial Analysis Gemini 3.7 Flash evaluation",
      "sourceUrl": "https://artificialanalysis.ai/models/comparisons/gemini-3-7-flash-vs-gemini-3-1-pro-preview",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    },
    {
      "id": "observation-chatbot-arena-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "chatbot-arena",
      "benchmarkVersion": "Chatbot Arena public leaderboard snapshot",
      "status": "available",
      "rawScore": 1490,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-arena",
      "sourceType": "benchmark_org",
      "sourceName": "Arena AI text leaderboard",
      "sourceUrl": "https://lmarena.ai/leaderboard/text",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Preliminary text Arena Elo with approximately ±8 Elo uncertainty."
    },
    {
      "id": "observation-livebench-gemini-3-7-flash-high",
      "fixture": false,
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "livebench",
      "benchmarkVersion": "LiveBench · Kimi K3 source snapshot",
      "status": "available",
      "rawScore": 78.8,
      "evaluatedAt": "2026-08-16T00:00:00.000Z",
      "firstPublishedAt": "2026-08-16T00:00:00.000Z",
      "lastVerifiedAt": "2026-08-16T00:00:00.000Z",
      "citationId": "citation-gemini-3-7-flash-livebench",
      "sourceType": "benchmark_org",
      "sourceName": "LiveBench Gemini 3.7 Flash leaderboard",
      "sourceUrl": "https://livebench.ai/",
      "modelVersion": "gemini-3.7-flash",
      "reasoningMode": "Gemini 3.7 Flash high reasoning",
      "reasoningBudget": "high effort",
      "comparability": "limited",
      "comparabilityNote": "Gemini 3.7 Flash high reasoning is retained as source-matched evidence; the public record does not establish one controlled evaluator and harness across the sheet.",
      "notes": "Value imported from the requested Gemini 3.7 Flash high-reasoning source-matched benchmark sheet row."
    }
  ],
  "alternateEvidence": [],
  "coverageNotes": [
    {
      "id": "coverage-gpqa-luna",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "The public GPQA Diamond snapshot did not expose a comparable GPT-5.6 Luna result. No value is inferred from GPT-5.6 Sol.",
      "citationId": "citation-aa-gpqa",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-prbench-grok-4-6",
      "modelId": "grok-4-6-high",
      "benchmarkId": "prbench-legal",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "Scale's public PRBench Legal leaderboard did not list Grok 4.6 at the snapshot date.",
      "citationId": "citation-prbench-legal",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdp-pdf-grok-4-6",
      "modelId": "grok-4-6-high",
      "benchmarkId": "gdp-pdf",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "The public GDP.pdf leaderboard did not list Grok 4.6 at the snapshot date.",
      "citationId": "citation-gdp-pdf",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Fable 5 · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Fable 5 · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Fable 5 · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Fable 5 · Extra High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Sol · None at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Sol · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Sol · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Sol · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Sol · Extra High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Terra · None at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Terra · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Terra · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Terra · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Terra · Extra High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Terra · Max at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Luna · None at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Luna · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Luna · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Luna · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GPT-5.6 Luna · Extra High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Grok 4.6 · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Grok 4.6 · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Grok 4.6 · Extra High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Kimi K3 · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Kimi K3 · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Gemini 3.1 Pro · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Gemini 3.1 Pro · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for DeepSeek V4 Pro · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for DeepSeek V4 Pro · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Qwen3.8-Max · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Qwen3.8-Max · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Muse Spark 1.2 · Minimal at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Muse Spark 1.2 · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Muse Spark 1.2 · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Muse Spark 1.2 · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GLM-5.2 · None at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GLM-5.2 · Minimal at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GLM-5.2 · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GLM-5.2 · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GLM-5.2 · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GLM-5.2 · Extra High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for GLM-5.2 · Max at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Gemini 3.7 Flash · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Gemini 3.7 Flash · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Sonnet 5 · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Sonnet 5 · Medium at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Sonnet 5 · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Sonnet 5 · Extra High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for Sonnet 5 · Max at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for DeepSeek V4 Flash · Low at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for DeepSeek V4 Flash · High at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gdpval-aa-v2-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "gdpval-aa-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GDPval-AA v2 result was verified for DeepSeek V4 Flash · Max at the AA Intelligence Index v4.1.1 · GDPval-AA v2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Fable 5 · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Fable 5 · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Fable 5 · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Fable 5 · Extra High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Sol · None at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Sol · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Sol · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Sol · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Sol · Extra High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Terra · None at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Terra · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Terra · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Terra · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Terra · Extra High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Terra · Max at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Luna · None at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Luna · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Luna · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Luna · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GPT-5.6 Luna · Extra High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Grok 4.6 · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Grok 4.6 · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Kimi K3 · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Kimi K3 · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Gemini 3.1 Pro · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Gemini 3.1 Pro · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for DeepSeek V4 Pro · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for DeepSeek V4 Pro · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Qwen3.8-Max · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Qwen3.8-Max · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Muse Spark 1.2 · Minimal at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Muse Spark 1.2 · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Muse Spark 1.2 · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Muse Spark 1.2 · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GLM-5.2 · None at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GLM-5.2 · Minimal at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GLM-5.2 · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GLM-5.2 · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GLM-5.2 · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GLM-5.2 · Extra High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for GLM-5.2 · Max at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Gemini 3.7 Flash · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Gemini 3.7 Flash · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Sonnet 5 · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Sonnet 5 · Medium at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Sonnet 5 · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Sonnet 5 · Extra High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for Sonnet 5 · Max at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for DeepSeek V4 Flash · Low at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for DeepSeek V4 Flash · High at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public GPQA Diamond result was verified for DeepSeek V4 Flash · Max at the AA Intelligence Index v4.1.1 · GPQA Diamond snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Opus 5 · Max at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Opus 5 · XHigh at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Opus 5 · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Opus 5 · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Opus 5 · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Fable 5 · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Fable 5 · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Fable 5 · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Fable 5 · Extra High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Fable 5 · Max at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Sol · None at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Sol · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Sol · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Sol · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Sol · Extra High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Sol · Max at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Terra · None at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Terra · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Terra · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Terra · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Terra · Extra High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Terra · Max at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Luna · None at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Luna · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Luna · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Luna · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Luna · Extra High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GPT-5.6 Luna · Max at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Grok 4.6 · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Grok 4.6 · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Grok 4.6 · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Grok 4.6 · Extra High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Kimi K3 · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Kimi K3 · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Gemini 3.1 Pro · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Gemini 3.1 Pro · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Gemini 3.1 Pro · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for DeepSeek V4 Pro · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for DeepSeek V4 Pro · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Qwen3.8-Max · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Qwen3.8-Max · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Muse Spark 1.2 · Minimal at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Muse Spark 1.2 · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Muse Spark 1.2 · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Muse Spark 1.2 · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GLM-5.2 · None at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GLM-5.2 · Minimal at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GLM-5.2 · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GLM-5.2 · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GLM-5.2 · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GLM-5.2 · Extra High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for GLM-5.2 · Max at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Gemini 3.7 Flash · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Gemini 3.7 Flash · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Sonnet 5 · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Sonnet 5 · Medium at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Sonnet 5 · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Sonnet 5 · Extra High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for Sonnet 5 · Max at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for DeepSeek V4 Flash · Low at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for DeepSeek V4 Flash · High at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Artificial Analysis Intelligence Index result was verified for DeepSeek V4 Flash · Max at the AA Intelligence Index v4.1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Opus 5 · Max at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Opus 5 · XHigh at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Opus 5 · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Opus 5 · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Fable 5 · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Fable 5 · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Fable 5 · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Sol · None at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Sol · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Sol · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Sol · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Sol · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Terra · None at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Terra · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Terra · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Terra · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Terra · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Terra · Max at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Luna · None at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Luna · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Luna · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Luna · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Luna · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GPT-5.6 Luna · Max at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Grok 4.6 · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Grok 4.6 · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Grok 4.6 · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Kimi K3 · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Kimi K3 · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Gemini 3.1 Pro · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Gemini 3.1 Pro · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Gemini 3.1 Pro · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for DeepSeek V4 Pro · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for DeepSeek V4 Pro · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for DeepSeek V4 Pro · Max at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Qwen3.8-Max · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Qwen3.8-Max · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Qwen3.8-Max · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Muse Spark 1.2 · Minimal at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Muse Spark 1.2 · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Muse Spark 1.2 · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Muse Spark 1.2 · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Muse Spark 1.2 · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GLM-5.2 · None at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GLM-5.2 · Minimal at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GLM-5.2 · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GLM-5.2 · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GLM-5.2 · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GLM-5.2 · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for GLM-5.2 · Max at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Gemini 3.7 Flash · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Gemini 3.7 Flash · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Gemini 3.7 Flash · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Sonnet 5 · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Sonnet 5 · Medium at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Sonnet 5 · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Sonnet 5 · Extra High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for Sonnet 5 · Max at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for DeepSeek V4 Flash · Low at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for DeepSeek V4 Flash · High at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public FrontierCode v1.1 Extended result was verified for DeepSeek V4 Flash · Max at the FrontierCode v1.1 Extended snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for GPT-5.6 Sol · None at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for GPT-5.6 Terra · None at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for GPT-5.6 Luna · None at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Gemini 3.1 Pro · Low at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Gemini 3.1 Pro · Medium at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Gemini 3.1 Pro · High at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for DeepSeek V4 Pro · Low at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for DeepSeek V4 Pro · High at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for DeepSeek V4 Pro · Max at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Qwen3.8-Max · Low at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Qwen3.8-Max · Medium at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Qwen3.8-Max · Extra High at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Muse Spark 1.2 · Minimal at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Muse Spark 1.2 · Low at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Muse Spark 1.2 · Medium at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Muse Spark 1.2 · High at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for Muse Spark 1.2 · Extra High at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for GLM-5.2 · None at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for GLM-5.2 · Minimal at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for GLM-5.2 · Low at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for GLM-5.2 · Medium at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for GLM-5.2 · Extra High at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for DeepSeek V4 Flash · Low at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for DeepSeek V4 Flash · High at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public CursorBench 3.2 result was verified for DeepSeek V4 Flash · Max at the CursorBench 3.2 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Opus 5 · XHigh at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Opus 5 · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Opus 5 · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Opus 5 · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Fable 5 · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Fable 5 · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Fable 5 · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Fable 5 · Extra High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Sol · None at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Sol · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Sol · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Sol · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Sol · Extra High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Terra · None at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Terra · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Terra · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Terra · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Terra · Extra High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Terra · Max at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Luna · None at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Luna · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Luna · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Luna · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GPT-5.6 Luna · Extra High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Grok 4.6 · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Grok 4.6 · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Grok 4.6 · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Kimi K3 · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Kimi K3 · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Gemini 3.1 Pro · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Gemini 3.1 Pro · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Gemini 3.1 Pro · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for DeepSeek V4 Pro · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for DeepSeek V4 Pro · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Qwen3.8-Max · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Qwen3.8-Max · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Muse Spark 1.2 · Minimal at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Muse Spark 1.2 · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Muse Spark 1.2 · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Muse Spark 1.2 · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GLM-5.2 · None at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GLM-5.2 · Minimal at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GLM-5.2 · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GLM-5.2 · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GLM-5.2 · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for GLM-5.2 · Extra High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Gemini 3.7 Flash · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Gemini 3.7 Flash · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Sonnet 5 · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Sonnet 5 · Medium at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Sonnet 5 · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for Sonnet 5 · Extra High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for DeepSeek V4 Flash · Low at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-deepswe-1-1-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "deepswe-1-1",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public DeepSWE v1.1 result was verified for DeepSeek V4 Flash · High at the DeepSWE v1.1 snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Opus 5 · XHigh at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Opus 5 · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Opus 5 · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Opus 5 · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Fable 5 · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Fable 5 · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Fable 5 · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Fable 5 · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Sol · None at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Sol · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Sol · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Sol · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Sol · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Terra · None at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Terra · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Terra · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Terra · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Terra · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Terra · Max at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Luna · None at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Luna · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Luna · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Luna · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Luna · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GPT-5.6 Luna · Max at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Grok 4.6 · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Grok 4.6 · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Grok 4.6 · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Kimi K3 · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Kimi K3 · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Gemini 3.1 Pro · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Gemini 3.1 Pro · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Gemini 3.1 Pro · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for DeepSeek V4 Pro · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for DeepSeek V4 Pro · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for DeepSeek V4 Pro · Max at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Qwen3.8-Max · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Qwen3.8-Max · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Qwen3.8-Max · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Muse Spark 1.2 · Minimal at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Muse Spark 1.2 · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Muse Spark 1.2 · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Muse Spark 1.2 · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Muse Spark 1.2 · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GLM-5.2 · None at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GLM-5.2 · Minimal at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GLM-5.2 · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GLM-5.2 · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GLM-5.2 · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GLM-5.2 · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for GLM-5.2 · Max at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Gemini 3.7 Flash · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Gemini 3.7 Flash · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Gemini 3.7 Flash · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Sonnet 5 · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Sonnet 5 · Medium at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Sonnet 5 · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Sonnet 5 · Extra High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for Sonnet 5 · Max at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for DeepSeek V4 Flash · Low at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for DeepSeek V4 Flash · High at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-Agents result was verified for DeepSeek V4 Flash · Max at the APEX-Agents public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Opus 5 · XHigh at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Opus 5 · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Opus 5 · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Opus 5 · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Fable 5 · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Fable 5 · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Fable 5 · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Fable 5 · Extra High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Sol · None at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Sol · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Sol · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Sol · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Sol · Extra High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Sol · Max at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Terra · None at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Terra · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Terra · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Terra · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Terra · Extra High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Terra · Max at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Luna · None at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Luna · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Luna · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Luna · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Luna · Extra High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GPT-5.6 Luna · Max at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Grok 4.6 · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Grok 4.6 · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Kimi K3 · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Kimi K3 · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Gemini 3.1 Pro · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Gemini 3.1 Pro · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Gemini 3.1 Pro · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for DeepSeek V4 Pro · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for DeepSeek V4 Pro · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for DeepSeek V4 Pro · Max at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Qwen3.8-Max · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Qwen3.8-Max · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Qwen3.8-Max · Extra High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Muse Spark 1.2 · Minimal at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Muse Spark 1.2 · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Muse Spark 1.2 · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Muse Spark 1.2 · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Muse Spark 1.2 · Extra High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GLM-5.2 · None at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GLM-5.2 · Minimal at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GLM-5.2 · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GLM-5.2 · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GLM-5.2 · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GLM-5.2 · Extra High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for GLM-5.2 · Max at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Gemini 3.7 Flash · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Gemini 3.7 Flash · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Gemini 3.7 Flash · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Sonnet 5 · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Sonnet 5 · Medium at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Sonnet 5 · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Sonnet 5 · Extra High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for Sonnet 5 · Max at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for DeepSeek V4 Flash · Low at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for DeepSeek V4 Flash · High at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public APEX-SWE result was verified for DeepSeek V4 Flash · Max at the APEX-SWE public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Opus 5 · Max at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Opus 5 · XHigh at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Opus 5 · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Opus 5 · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Opus 5 · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Fable 5 · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Fable 5 · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Fable 5 · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Fable 5 · Extra High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Sol · None at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Sol · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Sol · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Sol · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Sol · Extra High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Sol · Max at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Terra · None at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Terra · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Terra · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Terra · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Terra · Extra High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Terra · Max at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Luna · None at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Luna · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Luna · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Luna · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Luna · Extra High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GPT-5.6 Luna · Max at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Grok 4.6 · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Grok 4.6 · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Grok 4.6 · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Grok 4.6 · Extra High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Kimi K3 · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Kimi K3 · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Gemini 3.1 Pro · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Gemini 3.1 Pro · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Gemini 3.1 Pro · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for DeepSeek V4 Pro · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for DeepSeek V4 Pro · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for DeepSeek V4 Pro · Max at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Qwen3.8-Max · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Qwen3.8-Max · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Muse Spark 1.2 · Minimal at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Muse Spark 1.2 · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Muse Spark 1.2 · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Muse Spark 1.2 · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GLM-5.2 · None at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GLM-5.2 · Minimal at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GLM-5.2 · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GLM-5.2 · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GLM-5.2 · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GLM-5.2 · Extra High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for GLM-5.2 · Max at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Gemini 3.7 Flash · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Gemini 3.7 Flash · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Sonnet 5 · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Sonnet 5 · Medium at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Sonnet 5 · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Sonnet 5 · Extra High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for Sonnet 5 · Max at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for DeepSeek V4 Flash · Low at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for DeepSeek V4 Flash · High at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public AA-Briefcase result was verified for DeepSeek V4 Flash · Max at the AA-Briefcase public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Opus 5 · Max at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Opus 5 · XHigh at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Opus 5 · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Opus 5 · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Opus 5 · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Fable 5 · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Fable 5 · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Fable 5 · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Fable 5 · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Sol · None at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Sol · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Sol · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Sol · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Sol · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Sol · Max at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Terra · None at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Terra · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Terra · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Terra · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Terra · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Terra · Max at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Luna · None at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Luna · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Luna · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Luna · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Luna · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GPT-5.6 Luna · Max at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Grok 4.6 · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Grok 4.6 · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Grok 4.6 · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Grok 4.6 · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Kimi K3 · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Kimi K3 · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Gemini 3.1 Pro · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Gemini 3.1 Pro · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Gemini 3.1 Pro · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for DeepSeek V4 Pro · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for DeepSeek V4 Pro · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for DeepSeek V4 Pro · Max at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Qwen3.8-Max · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Qwen3.8-Max · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Qwen3.8-Max · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Muse Spark 1.2 · Minimal at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Muse Spark 1.2 · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Muse Spark 1.2 · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Muse Spark 1.2 · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Muse Spark 1.2 · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GLM-5.2 · None at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GLM-5.2 · Minimal at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GLM-5.2 · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GLM-5.2 · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GLM-5.2 · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GLM-5.2 · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for GLM-5.2 · Max at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Gemini 3.7 Flash · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Gemini 3.7 Flash · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Gemini 3.7 Flash · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Sonnet 5 · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Sonnet 5 · Medium at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Sonnet 5 · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Sonnet 5 · Extra High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for Sonnet 5 · Max at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for DeepSeek V4 Flash · Low at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for DeepSeek V4 Flash · High at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly readable public Harvey LAB-AA result was verified for DeepSeek V4 Flash · Max at the Harvey LAB-AA public leaderboard snapshot snapshot; this cell remains missing rather than inferred.",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-sol-none-8",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-sol-low-9",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-sol-medium-10",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-sol-high-11",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-sol-xhigh-12",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-terra-none-13",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-terra-low-14",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-terra-medium-15",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-terra-high-16",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-terra-xhigh-17",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-terra-max-18",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-luna-none-19",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-luna-low-20",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-luna-medium-21",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-luna-high-22",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-luna-xhigh-23",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gpt-5-6-luna-max-24",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-grok-4-6-low-25",
      "modelId": "grok-4-6-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-grok-4-6-medium-26",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-grok-4-6-high-27",
      "modelId": "grok-4-6-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-kimi-k3-low-28",
      "modelId": "kimi-k3-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-kimi-k3-high-29",
      "modelId": "kimi-k3-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gemini-3-1-pro-low-30",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gemini-3-1-pro-medium-31",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-deepseek-v4-pro-low-32",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-deepseek-v4-pro-high-33",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-qwen-3-8-max-low-34",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-qwen-3-8-max-medium-35",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-qwen-3-8-max-xhigh-36",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for qwen-3-8-max-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-muse-spark-1-2-minimal-37",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-muse-spark-1-2-low-38",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-muse-spark-1-2-medium-39",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-muse-spark-1-2-high-40",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-muse-spark-1-2-xhigh-41",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-glm-5-2-none-42",
      "modelId": "glm-5-2-none",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-glm-5-2-minimal-43",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-glm-5-2-low-44",
      "modelId": "glm-5-2-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-glm-5-2-medium-45",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-glm-5-2-high-46",
      "modelId": "glm-5-2-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-glm-5-2-xhigh-47",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-glm-5-2-max-48",
      "modelId": "glm-5-2-max",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gemini-3-7-flash-low-49",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gemini-3-7-flash-medium-50",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-gemini-3-7-flash-high-51",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-sonnet-5-low-52",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-sonnet-5-medium-53",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-sonnet-5-high-54",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-sonnet-5-xhigh-55",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-claude-sonnet-5-max-56",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-deepseek-v4-flash-low-57",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-deepseek-v4-flash-high-58",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-deepseek-v4-flash-max-59",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals table exposes the CLI split for the seven target model families, but it does not pin max/xhigh effort for every row. Rows are displayed as source evidence and remain excluded from the headline score until effort and serving configuration are pinned. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-sol-none-8",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-sol-low-9",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-sol-medium-10",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-sol-high-11",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-sol-xhigh-12",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-terra-none-13",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-terra-low-14",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-terra-medium-15",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-terra-high-16",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-terra-xhigh-17",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-terra-max-18",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-luna-none-19",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-luna-low-20",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-luna-medium-21",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-luna-high-22",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-luna-xhigh-23",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gpt-5-6-luna-max-24",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-grok-4-6-low-25",
      "modelId": "grok-4-6-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-grok-4-6-medium-26",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-grok-4-6-high-27",
      "modelId": "grok-4-6-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-kimi-k3-low-28",
      "modelId": "kimi-k3-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-kimi-k3-high-29",
      "modelId": "kimi-k3-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gemini-3-1-pro-low-30",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gemini-3-1-pro-medium-31",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-deepseek-v4-pro-low-32",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-deepseek-v4-pro-high-33",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-qwen-3-8-max-low-34",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-qwen-3-8-max-medium-35",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-qwen-3-8-max-xhigh-36",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for qwen-3-8-max-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-muse-spark-1-2-minimal-37",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-muse-spark-1-2-low-38",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-muse-spark-1-2-medium-39",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-muse-spark-1-2-high-40",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-muse-spark-1-2-xhigh-41",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-glm-5-2-none-42",
      "modelId": "glm-5-2-none",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-glm-5-2-minimal-43",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-glm-5-2-low-44",
      "modelId": "glm-5-2-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-glm-5-2-medium-45",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-glm-5-2-high-46",
      "modelId": "glm-5-2-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-glm-5-2-xhigh-47",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-glm-5-2-max-48",
      "modelId": "glm-5-2-max",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gemini-3-7-flash-low-49",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gemini-3-7-flash-medium-50",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-gemini-3-7-flash-high-51",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-sonnet-5-low-52",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-sonnet-5-medium-53",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-sonnet-5-high-54",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-sonnet-5-xhigh-55",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-claude-sonnet-5-max-56",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-deepseek-v4-flash-low-57",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-deepseek-v4-flash-high-58",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-deepseek-v4-flash-max-59",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The public page exposes the Dataroom Summaries category for all seven frontier model families. Because the page does not pin the requested max/xhigh reasoning setting and the tasks are private, the rows are displayed but excluded from the headline score. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-sol-none-8",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-sol-low-9",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-sol-medium-10",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-sol-high-11",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-sol-xhigh-12",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-terra-none-13",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-terra-low-14",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-terra-medium-15",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-terra-high-16",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-terra-xhigh-17",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-terra-max-18",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-luna-none-19",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-luna-low-20",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-luna-medium-21",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-luna-high-22",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-luna-xhigh-23",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gpt-5-6-luna-max-24",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-grok-4-6-low-25",
      "modelId": "grok-4-6-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-grok-4-6-medium-26",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-grok-4-6-high-27",
      "modelId": "grok-4-6-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-kimi-k3-low-28",
      "modelId": "kimi-k3-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-kimi-k3-high-29",
      "modelId": "kimi-k3-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gemini-3-1-pro-low-30",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gemini-3-1-pro-medium-31",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-deepseek-v4-pro-low-32",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-deepseek-v4-pro-high-33",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-qwen-3-8-max-low-34",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-qwen-3-8-max-medium-35",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-qwen-3-8-max-xhigh-36",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for qwen-3-8-max-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-muse-spark-1-2-minimal-37",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-muse-spark-1-2-low-38",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-muse-spark-1-2-medium-39",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-muse-spark-1-2-high-40",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-muse-spark-1-2-xhigh-41",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-glm-5-2-none-42",
      "modelId": "glm-5-2-none",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-glm-5-2-minimal-43",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-glm-5-2-low-44",
      "modelId": "glm-5-2-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-glm-5-2-medium-45",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-glm-5-2-high-46",
      "modelId": "glm-5-2-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-glm-5-2-xhigh-47",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-glm-5-2-max-48",
      "modelId": "glm-5-2-max",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gemini-3-7-flash-low-49",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gemini-3-7-flash-medium-50",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-gemini-3-7-flash-high-51",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-sonnet-5-low-52",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-sonnet-5-medium-53",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-sonnet-5-high-54",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-sonnet-5-xhigh-55",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-claude-sonnet-5-max-56",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-deepseek-v4-flash-low-57",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-deepseek-v4-flash-high-58",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-deepseek-v4-flash-max-59",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The evaluated-model catalog and Health practice-area heatmap include all seven target families, but the public page does not pin max/xhigh reasoning or expose a versioned export of the overall aggregate. The native slice is therefore display-only. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-opus-5-max-0",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-opus-5-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-opus-5-xhigh-1",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-opus-5-high-2",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-opus-5-medium-3",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-opus-5-low-4",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-fable-5-low-5",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-fable-5-medium-6",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-fable-5-high-7",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-fable-5-xhigh-8",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-sol-none-9",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-sol-low-10",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-sol-medium-11",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-sol-high-12",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-sol-xhigh-13",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-sol-max-14",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-sol-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-terra-none-15",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-terra-low-16",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-terra-medium-17",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-terra-high-18",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-terra-xhigh-19",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-terra-max-20",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-luna-none-21",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-luna-low-22",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-luna-medium-23",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-luna-high-24",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-luna-xhigh-25",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gpt-5-6-luna-max-26",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-grok-4-6-low-27",
      "modelId": "grok-4-6-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-grok-4-6-medium-28",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-grok-4-6-high-29",
      "modelId": "grok-4-6-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-kimi-k3-low-30",
      "modelId": "kimi-k3-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-kimi-k3-high-31",
      "modelId": "kimi-k3-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gemini-3-1-pro-low-32",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gemini-3-1-pro-medium-33",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gemini-3-1-pro-high-34",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gemini-3-1-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-deepseek-v4-pro-low-35",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-deepseek-v4-pro-high-36",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-qwen-3-8-max-low-38",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-qwen-3-8-max-medium-39",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-muse-spark-1-2-minimal-41",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-muse-spark-1-2-low-42",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-muse-spark-1-2-medium-43",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-muse-spark-1-2-high-44",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-glm-5-2-none-46",
      "modelId": "glm-5-2-none",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-glm-5-2-minimal-47",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-glm-5-2-low-48",
      "modelId": "glm-5-2-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-glm-5-2-medium-49",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-glm-5-2-high-50",
      "modelId": "glm-5-2-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-glm-5-2-xhigh-51",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-glm-5-2-max-52",
      "modelId": "glm-5-2-max",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gemini-3-7-flash-low-53",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-gemini-3-7-flash-medium-54",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-sonnet-5-low-56",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-sonnet-5-medium-57",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-sonnet-5-high-58",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-sonnet-5-xhigh-59",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-claude-sonnet-5-max-60",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-deepseek-v4-flash-low-61",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-deepseek-v4-flash-high-62",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-deepseek-v4-flash-max-63",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark · Vals run score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals model catalog lists all seven target families, but only three target rows were directly readable in the captured page. The DeepSeek row visible on the page was not the requested V4 Pro configuration and is not substituted. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-sol-none-8",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-sol-low-9",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-sol-medium-10",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-sol-high-11",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-sol-xhigh-12",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-sol-max-13",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-sol-max in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-terra-none-14",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-terra-low-15",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-terra-medium-16",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-terra-high-17",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-terra-xhigh-18",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-terra-max-19",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-luna-none-20",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-luna-low-21",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-luna-medium-22",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-luna-high-23",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-luna-xhigh-24",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gpt-5-6-luna-max-25",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-grok-4-6-low-26",
      "modelId": "grok-4-6-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-grok-4-6-medium-27",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-grok-4-6-high-28",
      "modelId": "grok-4-6-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-grok-4-6-xhigh-29",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for grok-4-6-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-kimi-k3-low-30",
      "modelId": "kimi-k3-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-kimi-k3-high-31",
      "modelId": "kimi-k3-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-kimi-k3-max-32",
      "modelId": "kimi-k3-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for kimi-k3-max in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gemini-3-1-pro-low-33",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gemini-3-1-pro-medium-34",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gemini-3-1-pro-high-35",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gemini-3-1-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-deepseek-v4-pro-low-36",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-deepseek-v4-pro-high-37",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-deepseek-v4-pro-max-38",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for deepseek-v4-pro-max in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-qwen-3-8-max-low-39",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-qwen-3-8-max-medium-40",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-qwen-3-8-max-xhigh-41",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for qwen-3-8-max-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-muse-spark-1-2-minimal-42",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-muse-spark-1-2-low-43",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-muse-spark-1-2-medium-44",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-muse-spark-1-2-high-45",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-muse-spark-1-2-xhigh-46",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-glm-5-2-none-47",
      "modelId": "glm-5-2-none",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-glm-5-2-minimal-48",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-glm-5-2-low-49",
      "modelId": "glm-5-2-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-glm-5-2-medium-50",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-glm-5-2-high-51",
      "modelId": "glm-5-2-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-glm-5-2-xhigh-52",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-glm-5-2-max-53",
      "modelId": "glm-5-2-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gemini-3-7-flash-low-54",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gemini-3-7-flash-medium-55",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-gemini-3-7-flash-high-56",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-sonnet-5-low-57",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-sonnet-5-medium-58",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-sonnet-5-high-59",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-sonnet-5-xhigh-60",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-claude-sonnet-5-max-61",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-deepseek-v4-flash-low-62",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-deepseek-v4-flash-high-63",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-deepseek-v4-flash-max-64",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vibe Code Bench v1.1 score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The public leaderboard catalog includes all seven target families, but only the Opus 5 and Fable 5 score rows were directly readable in the captured v1.1 snapshot. No values are copied from similarly named Vibe Code or coding benchmarks. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-sol-none-9",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-sol-low-10",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-sol-medium-11",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-sol-high-12",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-sol-xhigh-13",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-terra-none-14",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-terra-low-15",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-terra-medium-16",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-terra-high-17",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-terra-xhigh-18",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-terra-max-19",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-luna-none-20",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-luna-low-21",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-luna-medium-22",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-luna-high-23",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-luna-xhigh-24",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gpt-5-6-luna-max-25",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-grok-4-6-low-26",
      "modelId": "grok-4-6-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-grok-4-6-medium-27",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-grok-4-6-high-28",
      "modelId": "grok-4-6-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-kimi-k3-low-30",
      "modelId": "kimi-k3-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-kimi-k3-high-31",
      "modelId": "kimi-k3-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gemini-3-1-pro-low-32",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gemini-3-1-pro-medium-33",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gemini-3-1-pro-high-34",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gemini-3-1-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-deepseek-v4-pro-low-35",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-deepseek-v4-pro-high-36",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-qwen-3-8-max-low-37",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-qwen-3-8-max-medium-38",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-muse-spark-1-2-minimal-40",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-muse-spark-1-2-low-41",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-muse-spark-1-2-medium-42",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-muse-spark-1-2-high-43",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-glm-5-2-none-45",
      "modelId": "glm-5-2-none",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-glm-5-2-minimal-46",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-glm-5-2-low-47",
      "modelId": "glm-5-2-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-glm-5-2-medium-48",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-glm-5-2-high-49",
      "modelId": "glm-5-2-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-glm-5-2-xhigh-50",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-glm-5-2-max-51",
      "modelId": "glm-5-2-max",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gemini-3-7-flash-low-52",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gemini-3-7-flash-medium-53",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-gemini-3-7-flash-high-54",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-sonnet-5-low-55",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-sonnet-5-medium-56",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-sonnet-5-high-57",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-sonnet-5-xhigh-58",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-claude-sonnet-5-max-59",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-deepseek-v4-flash-low-60",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-deepseek-v4-flash-high-61",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-deepseek-v4-flash-max-62",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SWE-bench Verified · Vals run score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals leaderboard explicitly lists all seven target families, but the captured row-level export exposed only four target scores. Vals does not consistently publish the requested max/xhigh effort for its provider-default rows, so no absent row is reconstructed from another SWE-bench harness. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-sol-none-9",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-sol-low-10",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-sol-medium-11",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-sol-high-12",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-sol-xhigh-13",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-terra-none-14",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-terra-low-15",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-terra-medium-16",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-terra-high-17",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-terra-xhigh-18",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-terra-max-19",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-luna-none-20",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-luna-low-21",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-luna-medium-22",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-luna-high-23",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-luna-xhigh-24",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gpt-5-6-luna-max-25",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-grok-4-6-low-26",
      "modelId": "grok-4-6-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-grok-4-6-medium-27",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-grok-4-6-high-28",
      "modelId": "grok-4-6-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-kimi-k3-low-30",
      "modelId": "kimi-k3-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-kimi-k3-high-31",
      "modelId": "kimi-k3-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gemini-3-1-pro-low-33",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gemini-3-1-pro-medium-34",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-deepseek-v4-pro-low-35",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-deepseek-v4-pro-high-36",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-deepseek-v4-pro-max-37",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for deepseek-v4-pro-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-qwen-3-8-max-low-38",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-qwen-3-8-max-medium-39",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-muse-spark-1-2-minimal-41",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-muse-spark-1-2-low-42",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-muse-spark-1-2-medium-43",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-muse-spark-1-2-high-44",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-glm-5-2-none-46",
      "modelId": "glm-5-2-none",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-glm-5-2-minimal-47",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-glm-5-2-low-48",
      "modelId": "glm-5-2-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-glm-5-2-medium-49",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-glm-5-2-high-50",
      "modelId": "glm-5-2-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-glm-5-2-xhigh-51",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-glm-5-2-max-52",
      "modelId": "glm-5-2-max",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gemini-3-7-flash-low-53",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gemini-3-7-flash-medium-54",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-gemini-3-7-flash-high-55",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-sonnet-5-low-56",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-sonnet-5-medium-57",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-sonnet-5-high-58",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-sonnet-5-xhigh-59",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-claude-sonnet-5-max-60",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-deepseek-v4-flash-low-61",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-deepseek-v4-flash-high-62",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-deepseek-v4-flash-max-63",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMLU-Pro · Vals run score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The Vals page reports a broad evaluated-model catalog, but the complete seven-model row set was not exposed in the captured public export. Only directly readable Opus 5, GPT-5.6 Sol, and Gemini 3.1 Pro rows are ingested. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-sol-none-8",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-sol-low-9",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-sol-medium-10",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-sol-high-11",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-sol-xhigh-12",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-terra-none-13",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-terra-low-14",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-terra-medium-15",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-terra-high-16",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-terra-xhigh-17",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-terra-max-18",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-luna-none-19",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-luna-low-20",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-luna-medium-21",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-luna-high-22",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-luna-xhigh-23",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gpt-5-6-luna-max-24",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-grok-4-6-low-25",
      "modelId": "grok-4-6-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-grok-4-6-medium-26",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-grok-4-6-high-27",
      "modelId": "grok-4-6-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-grok-4-6-xhigh-28",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for grok-4-6-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-kimi-k3-low-29",
      "modelId": "kimi-k3-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-kimi-k3-high-30",
      "modelId": "kimi-k3-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gemini-3-1-pro-low-32",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gemini-3-1-pro-medium-33",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gemini-3-1-pro-high-34",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gemini-3-1-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-deepseek-v4-pro-low-35",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-deepseek-v4-pro-high-36",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-deepseek-v4-pro-max-37",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for deepseek-v4-pro-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-qwen-3-8-max-low-38",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-qwen-3-8-max-medium-39",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-muse-spark-1-2-minimal-41",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-muse-spark-1-2-low-42",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-muse-spark-1-2-medium-43",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-muse-spark-1-2-high-44",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-muse-spark-1-2-xhigh-45",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-glm-5-2-none-46",
      "modelId": "glm-5-2-none",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-glm-5-2-minimal-47",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-glm-5-2-low-48",
      "modelId": "glm-5-2-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-glm-5-2-medium-49",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-glm-5-2-high-50",
      "modelId": "glm-5-2-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-glm-5-2-xhigh-51",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-glm-5-2-max-52",
      "modelId": "glm-5-2-max",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gemini-3-7-flash-low-53",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gemini-3-7-flash-medium-54",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-gemini-3-7-flash-high-55",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-sonnet-5-low-56",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-sonnet-5-medium-57",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-sonnet-5-high-58",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-sonnet-5-xhigh-59",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-claude-sonnet-5-max-60",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-deepseek-v4-flash-low-61",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-deepseek-v4-flash-high-62",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-deepseek-v4-flash-max-63",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MMMU-Pro · Vals run score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The public Vals page exposes Opus 5, Fable 5, and GPT-5.6 Sol rows in the captured snapshot. Other target families remain explicitly missing because a comparable Vals row was not directly verified. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-sol-none-8",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-sol-low-9",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-sol-medium-10",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-sol-high-11",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-sol-xhigh-12",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-terra-none-13",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-terra-low-14",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-terra-medium-15",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-terra-high-16",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-terra-xhigh-17",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-terra-max-18",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-luna-none-19",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-luna-low-20",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-luna-medium-21",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-luna-high-22",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-luna-xhigh-23",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gpt-5-6-luna-max-24",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-grok-4-6-low-25",
      "modelId": "grok-4-6-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-grok-4-6-medium-26",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-grok-4-6-high-27",
      "modelId": "grok-4-6-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-kimi-k3-low-29",
      "modelId": "kimi-k3-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-kimi-k3-high-30",
      "modelId": "kimi-k3-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gemini-3-1-pro-low-32",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gemini-3-1-pro-medium-33",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-deepseek-v4-pro-low-34",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-deepseek-v4-pro-high-35",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-qwen-3-8-max-low-37",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-qwen-3-8-max-medium-38",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-muse-spark-1-2-minimal-40",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-muse-spark-1-2-low-41",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-muse-spark-1-2-medium-42",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-muse-spark-1-2-high-43",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-muse-spark-1-2-xhigh-44",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-glm-5-2-none-45",
      "modelId": "glm-5-2-none",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-glm-5-2-minimal-46",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-glm-5-2-low-47",
      "modelId": "glm-5-2-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-glm-5-2-medium-48",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-glm-5-2-high-49",
      "modelId": "glm-5-2-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-glm-5-2-xhigh-50",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-glm-5-2-max-51",
      "modelId": "glm-5-2-max",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gemini-3-7-flash-low-52",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gemini-3-7-flash-medium-53",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-gemini-3-7-flash-high-54",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-sonnet-5-low-55",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-sonnet-5-medium-56",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-sonnet-5-high-57",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-sonnet-5-xhigh-58",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-claude-sonnet-5-max-59",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-deepseek-v4-flash-low-60",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-deepseek-v4-flash-high-61",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-deepseek-v4-flash-max-62",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveCodeBench · Vals run score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The captured Vals page exposes Opus 5, Fable 5, GPT-5.6 Sol, and Gemini 3.1 Pro. A visible DeepSeek V4 row is not silently promoted to DeepSeek V4 Pro, and Grok 4.6 and Kimi K3 remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-opus-5-xhigh-0",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-opus-5-high-1",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-opus-5-medium-2",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-opus-5-low-3",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-fable-5-low-4",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-fable-5-medium-5",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-fable-5-high-6",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-fable-5-xhigh-7",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-sol-none-8",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-sol-low-9",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-sol-medium-10",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-sol-high-11",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-sol-xhigh-12",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-terra-none-13",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-terra-low-14",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-terra-medium-15",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-terra-high-16",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-terra-xhigh-17",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-terra-max-18",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-luna-none-19",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-luna-low-20",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-luna-medium-21",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-luna-high-22",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-luna-xhigh-23",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gpt-5-6-luna-max-24",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-grok-4-6-low-25",
      "modelId": "grok-4-6-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-grok-4-6-medium-26",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-grok-4-6-high-27",
      "modelId": "grok-4-6-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-grok-4-6-xhigh-28",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for grok-4-6-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-kimi-k3-low-29",
      "modelId": "kimi-k3-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-kimi-k3-high-30",
      "modelId": "kimi-k3-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gemini-3-1-pro-low-31",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gemini-3-1-pro-medium-32",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gemini-3-1-pro-high-33",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gemini-3-1-pro-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-deepseek-v4-pro-low-34",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-deepseek-v4-pro-high-35",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-deepseek-v4-pro-max-36",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for deepseek-v4-pro-max in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-qwen-3-8-max-low-37",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-qwen-3-8-max-medium-38",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-qwen-3-8-max-xhigh-39",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for qwen-3-8-max-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-muse-spark-1-2-minimal-40",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-muse-spark-1-2-low-41",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-muse-spark-1-2-medium-42",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-muse-spark-1-2-high-43",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-muse-spark-1-2-xhigh-44",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-glm-5-2-none-45",
      "modelId": "glm-5-2-none",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-glm-5-2-minimal-46",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-glm-5-2-low-47",
      "modelId": "glm-5-2-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-glm-5-2-medium-48",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-glm-5-2-high-49",
      "modelId": "glm-5-2-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-glm-5-2-xhigh-50",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-glm-5-2-max-51",
      "modelId": "glm-5-2-max",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gemini-3-7-flash-low-52",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gemini-3-7-flash-medium-53",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-gemini-3-7-flash-high-54",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-sonnet-5-low-55",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-sonnet-5-medium-56",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-sonnet-5-high-57",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-sonnet-5-xhigh-58",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-claude-sonnet-5-max-59",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-deepseek-v4-flash-low-60",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-deepseek-v4-flash-high-61",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-deepseek-v4-flash-max-62",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ProgramBench · Vals run score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. Only Opus 5, Fable 5, GPT-5.6 Sol, and Kimi K3 rows were directly readable in the captured Vals page. No value is inferred for Grok, Gemini, or DeepSeek from a different programming benchmark. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "The public track shows GPT-5.6 Sol XHigh at 60.0%; it is not substituted for the required Sol Max cell. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "The public track shows Grok 4.6 High at 62.7%; it is not substituted for the required Grok XHigh cell. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for qwen-3-8-max-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Integration score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-opus-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-opus-5-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-opus-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-opus-5-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-fable-5-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-fable-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-fable-5-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-fable-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-sol-none in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-sol-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-sol-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-sol-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-sol-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "The public track shows GPT-5.6 Sol XHigh at 31.5%; it is not substituted for the required Sol Max cell. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-terra-none in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-terra-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-terra-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-terra-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-terra-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-terra-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-luna-none in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-luna-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-luna-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-luna-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-luna-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gpt-5-6-luna-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for grok-4-6-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for grok-4-6-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for grok-4-6-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "The public track shows Grok 4.6 High at 50.0%; it is not substituted for the required Grok XHigh cell. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for kimi-k3-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for kimi-k3-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gemini-3-1-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gemini-3-1-pro-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for deepseek-v4-pro-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for deepseek-v4-pro-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for qwen-3-8-max-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for qwen-3-8-max-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for qwen-3-8-max-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for muse-spark-1-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for muse-spark-1-2-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for muse-spark-1-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for muse-spark-1-2-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for muse-spark-1-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for glm-5-2-none in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for glm-5-2-minimal in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for glm-5-2-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for glm-5-2-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for glm-5-2-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for glm-5-2-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for glm-5-2-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gemini-3-7-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gemini-3-7-flash-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for gemini-3-7-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-sonnet-5-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-sonnet-5-medium in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-sonnet-5-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-sonnet-5-xhigh in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for claude-sonnet-5-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for deepseek-v4-flash-low in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for deepseek-v4-flash-high in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified APEX-SWE · Observability score was ingested for deepseek-v4-flash-max in the 2026-08-15T00:00:00.000Z snapshot. The track is kept separate from the aggregate APEX-SWE record; no proxy or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiermath-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "frontiermath",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierMath v2 Tier 4 score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiermath",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierSWE score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierSWE score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierSWE score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierSWE score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierSWE score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontierswe-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "frontierswe",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierSWE score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified JobBench score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified JobBench score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified JobBench score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified JobBench score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified JobBench score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified JobBench score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BabyVision (with CI) score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BabyVision (with CI) score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BabyVision (with CI) score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BabyVision (with CI) score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-charxiv-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "charxiv",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CharXiv (with CI / RQ) score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-charxiv",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PerceptionBench score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PerceptionBench score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PerceptionBench score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PerceptionBench score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PerceptionBench score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PerceptionBench score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld-Verified score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld-Verified score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld-Verified score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld-Verified score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BrowseComp score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BrowseComp score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BrowseComp score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BrowseComp score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for qwen-3-8-max-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-browsecomp-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "browsecomp",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified BrowseComp score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-browsecomp",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Toolathlon-Verified score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Toolathlon-Verified score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-toolathlon-verified-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "toolathlon-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Toolathlon-Verified score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-humanitys-last-exam-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "humanitys-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Humanity's Last Exam (no tools) score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-humanitys-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified τ²-Bench Telecom score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified τ²-Bench Telecom score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for qwen-3-8-max-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-tau2-bench-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "tau2-bench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified τ²-Bench Telecom score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tau2-bench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-Omniscience score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-Omniscience score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-Omniscience score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for qwen-3-8-max-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-omniscience-index-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "aa-omniscience-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-Omniscience score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-omniscience",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-LCR score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-LCR score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-LCR score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AA-LCR score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ARC-AGI-2 score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ARC-AGI-2 score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for qwen-3-8-max-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-arc-agi-2-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "arc-agi-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ARC-AGI-2 score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-arc-prize",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-chatbot-arena-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "chatbot-arena",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LMArena (Chatbot Arena) — Text score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-chatbot-arena",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LiveBench score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LiveBench score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LiveBench score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LiveBench score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LiveBench score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Creative Writing v3 score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Creative Writing v3 score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Creative Writing v3 score was ingested for deepseek-v4-pro-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for qwen-3-8-max-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for gemini-3-7-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Creative Writing v3 score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-opus-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-opus-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-opus-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-opus-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-opus-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-fable-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-fable-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-fable-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-fable-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-fable-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-sol-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-sol-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-sol-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-sol-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-sol-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-sol-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-terra-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-terra-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-terra-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-terra-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-terra-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-terra-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-luna-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-luna-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-luna-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-luna-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-luna-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gpt-5-6-luna-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for grok-4-6-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for grok-4-6-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for grok-4-6-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Agents' Last Exam score was ingested for grok-4-6-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for kimi-k3-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for kimi-k3-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gemini-3-1-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gemini-3-1-pro-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Agents' Last Exam score was ingested for gemini-3-1-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for deepseek-v4-pro-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for deepseek-v4-pro-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for qwen-3-8-max-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for qwen-3-8-max-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for muse-spark-1-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for muse-spark-1-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for muse-spark-1-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for muse-spark-1-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for muse-spark-1-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for glm-5-2-none. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for glm-5-2-minimal. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for glm-5-2-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for glm-5-2-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for glm-5-2-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for glm-5-2-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for glm-5-2-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gemini-3-7-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for gemini-3-7-flash-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-sonnet-5-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-sonnet-5-medium. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-sonnet-5-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-sonnet-5-xhigh. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for claude-sonnet-5-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for deepseek-v4-flash-low. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for deepseek-v4-flash-high. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-agents-last-exam-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "agents-last-exam",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Agents' Last Exam score was ingested for deepseek-v4-flash-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-snorkel-agents-last-exam",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-opus-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-opus-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-opus-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-opus-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-opus-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-fable-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-fable-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-fable-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-fable-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-fable-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-sol-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-sol-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-sol-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-sol-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-sol-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-sol-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-terra-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-terra-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-terra-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-terra-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-terra-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-terra-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-luna-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-luna-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-luna-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-luna-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-luna-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gpt-5-6-luna-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for grok-4-6-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for grok-4-6-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for grok-4-6-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for grok-4-6-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for kimi-k3-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for kimi-k3-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for kimi-k3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gemini-3-1-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gemini-3-1-pro-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gemini-3-1-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for deepseek-v4-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for deepseek-v4-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for deepseek-v4-pro-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for qwen-3-8-max-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for qwen-3-8-max-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for muse-spark-1-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for muse-spark-1-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for muse-spark-1-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for muse-spark-1-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for muse-spark-1-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for glm-5-2-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for glm-5-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for glm-5-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for glm-5-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for glm-5-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for glm-5-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for glm-5-2-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gemini-3-7-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gemini-3-7-flash-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for gemini-3-7-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-sonnet-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-sonnet-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-sonnet-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-sonnet-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for claude-sonnet-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for deepseek-v4-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for deepseek-v4-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for deepseek-v4-flash-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-opus-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-opus-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-opus-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-opus-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-opus-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-fable-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-fable-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-fable-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-fable-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-fable-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-sol-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-sol-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-sol-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-sol-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-sol-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-sol-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-terra-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-terra-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-terra-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-terra-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-terra-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-terra-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-luna-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-luna-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-luna-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-luna-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-luna-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gpt-5-6-luna-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for grok-4-6-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for grok-4-6-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for grok-4-6-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for grok-4-6-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for kimi-k3-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for kimi-k3-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for kimi-k3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gemini-3-1-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gemini-3-1-pro-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gemini-3-1-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for deepseek-v4-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for deepseek-v4-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for deepseek-v4-pro-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for qwen-3-8-max-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for qwen-3-8-max-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for muse-spark-1-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for muse-spark-1-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for muse-spark-1-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for muse-spark-1-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for muse-spark-1-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for glm-5-2-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for glm-5-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for glm-5-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for glm-5-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for glm-5-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for glm-5-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for glm-5-2-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gemini-3-7-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gemini-3-7-flash-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for gemini-3-7-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-sonnet-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-sonnet-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-sonnet-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-sonnet-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for claude-sonnet-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for deepseek-v4-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for deepseek-v4-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for deepseek-v4-flash-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CoWorkBench score was ingested for claude-opus-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-opus-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-opus-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-opus-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-opus-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-fable-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-fable-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-fable-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-fable-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CoWorkBench score was ingested for claude-fable-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-sol-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-sol-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-sol-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-sol-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-sol-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-sol-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-terra-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-terra-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-terra-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-terra-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-terra-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-terra-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-luna-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-luna-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-luna-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-luna-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-luna-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gpt-5-6-luna-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for grok-4-6-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for grok-4-6-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for grok-4-6-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CoWorkBench score was ingested for grok-4-6-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for kimi-k3-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for kimi-k3-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CoWorkBench score was ingested for kimi-k3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gemini-3-1-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gemini-3-1-pro-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CoWorkBench score was ingested for gemini-3-1-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for deepseek-v4-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for deepseek-v4-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CoWorkBench score was ingested for deepseek-v4-pro-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for qwen-3-8-max-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for qwen-3-8-max-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for muse-spark-1-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for muse-spark-1-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for muse-spark-1-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for muse-spark-1-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for muse-spark-1-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for glm-5-2-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for glm-5-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for glm-5-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for glm-5-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for glm-5-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for glm-5-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for glm-5-2-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gemini-3-7-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gemini-3-7-flash-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for gemini-3-7-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-sonnet-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-sonnet-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-sonnet-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-sonnet-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for claude-sonnet-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for deepseek-v4-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for deepseek-v4-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CoWorkBench score was ingested for deepseek-v4-flash-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ERQA score was ingested for claude-opus-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-opus-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-opus-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-opus-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-opus-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-fable-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-fable-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-fable-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-fable-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ERQA score was ingested for claude-fable-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-sol-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-sol-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-sol-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-sol-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-sol-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-sol-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-terra-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-terra-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-terra-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-terra-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-terra-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-terra-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-luna-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-luna-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-luna-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-luna-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-luna-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gpt-5-6-luna-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for grok-4-6-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for grok-4-6-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for grok-4-6-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ERQA score was ingested for grok-4-6-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for kimi-k3-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for kimi-k3-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ERQA score was ingested for kimi-k3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gemini-3-1-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gemini-3-1-pro-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ERQA score was ingested for gemini-3-1-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for deepseek-v4-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for deepseek-v4-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ERQA score was ingested for deepseek-v4-pro-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for qwen-3-8-max-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for qwen-3-8-max-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for muse-spark-1-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for muse-spark-1-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for muse-spark-1-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for muse-spark-1-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for muse-spark-1-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for glm-5-2-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for glm-5-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for glm-5-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for glm-5-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for glm-5-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for glm-5-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for glm-5-2-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gemini-3-7-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gemini-3-7-flash-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for gemini-3-7-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-sonnet-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-sonnet-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-sonnet-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-sonnet-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for claude-sonnet-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for deepseek-v4-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for deepseek-v4-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ERQA score was ingested for deepseek-v4-flash-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-opus-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-opus-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-opus-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-opus-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-opus-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-fable-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-fable-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-fable-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-fable-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-fable-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-sol-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-sol-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-sol-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-sol-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-sol-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-sol-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-terra-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-terra-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-terra-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-terra-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-terra-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-terra-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-luna-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-luna-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-luna-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-luna-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-luna-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gpt-5-6-luna-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for grok-4-6-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for grok-4-6-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for grok-4-6-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LVBench (with Memory) score was ingested for grok-4-6-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for kimi-k3-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for kimi-k3-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LVBench (with Memory) score was ingested for kimi-k3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gemini-3-1-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gemini-3-1-pro-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LVBench (with Memory) score was ingested for gemini-3-1-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for deepseek-v4-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for deepseek-v4-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LVBench (with Memory) score was ingested for deepseek-v4-pro-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for qwen-3-8-max-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for qwen-3-8-max-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for muse-spark-1-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for muse-spark-1-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for muse-spark-1-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for muse-spark-1-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for muse-spark-1-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for glm-5-2-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for glm-5-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for glm-5-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for glm-5-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for glm-5-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for glm-5-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for glm-5-2-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gemini-3-7-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for gemini-3-7-flash-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-sonnet-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-sonnet-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-sonnet-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-sonnet-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for claude-sonnet-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for deepseek-v4-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for deepseek-v4-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LVBench (with Memory) score was ingested for deepseek-v4-flash-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-opus-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-opus-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-opus-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-opus-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-opus-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-fable-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-fable-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-fable-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-fable-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-fable-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-sol-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-sol-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-sol-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-sol-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-sol-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-sol-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-terra-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-terra-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-terra-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-terra-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-terra-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-terra-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-luna-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-luna-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-luna-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-luna-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-luna-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gpt-5-6-luna-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for grok-4-6-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for grok-4-6-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for grok-4-6-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for grok-4-6-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for kimi-k3-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for kimi-k3-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for kimi-k3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gemini-3-1-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gemini-3-1-pro-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gemini-3-1-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for deepseek-v4-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for deepseek-v4-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for deepseek-v4-pro-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for qwen-3-8-max-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for qwen-3-8-max-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for muse-spark-1-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for muse-spark-1-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for muse-spark-1-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for muse-spark-1-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for muse-spark-1-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for glm-5-2-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for glm-5-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for glm-5-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for glm-5-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for glm-5-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for glm-5-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for glm-5-2-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gemini-3-7-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gemini-3-7-flash-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for gemini-3-7-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-sonnet-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-sonnet-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-sonnet-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-sonnet-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for claude-sonnet-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for deepseek-v4-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for deepseek-v4-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for deepseek-v4-flash-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MobileWorld score was ingested for claude-opus-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-opus-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-opus-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-opus-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-opus-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-fable-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-fable-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-fable-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-fable-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MobileWorld score was ingested for claude-fable-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-sol-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-sol-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-sol-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-sol-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-sol-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-sol-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-terra-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-terra-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-terra-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-terra-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-terra-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-terra-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-luna-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-luna-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-luna-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-luna-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-luna-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gpt-5-6-luna-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for grok-4-6-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for grok-4-6-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for grok-4-6-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MobileWorld score was ingested for grok-4-6-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for kimi-k3-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for kimi-k3-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MobileWorld score was ingested for kimi-k3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gemini-3-1-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gemini-3-1-pro-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MobileWorld score was ingested for gemini-3-1-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for deepseek-v4-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for deepseek-v4-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MobileWorld score was ingested for deepseek-v4-pro-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for qwen-3-8-max-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for qwen-3-8-max-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for muse-spark-1-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for muse-spark-1-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for muse-spark-1-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for muse-spark-1-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for muse-spark-1-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for glm-5-2-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for glm-5-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for glm-5-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for glm-5-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for glm-5-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for glm-5-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for glm-5-2-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gemini-3-7-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gemini-3-7-flash-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for gemini-3-7-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-sonnet-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-sonnet-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-sonnet-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-sonnet-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for claude-sonnet-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for deepseek-v4-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for deepseek-v4-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MobileWorld score was ingested for deepseek-v4-flash-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified WebArena-Verified score was ingested for claude-opus-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-opus-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-opus-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-opus-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-opus-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-fable-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-fable-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-fable-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-fable-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified WebArena-Verified score was ingested for claude-fable-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-sol-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-sol-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-sol-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-sol-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-sol-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-sol-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-terra-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-terra-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-terra-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-terra-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-terra-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-terra-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-luna-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-luna-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-luna-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-luna-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-luna-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gpt-5-6-luna-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for grok-4-6-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for grok-4-6-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for grok-4-6-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified WebArena-Verified score was ingested for grok-4-6-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for kimi-k3-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for kimi-k3-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified WebArena-Verified score was ingested for kimi-k3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gemini-3-1-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gemini-3-1-pro-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified WebArena-Verified score was ingested for gemini-3-1-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for deepseek-v4-pro-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for deepseek-v4-pro-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified WebArena-Verified score was ingested for deepseek-v4-pro-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for qwen-3-8-max-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for qwen-3-8-max-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for muse-spark-1-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for muse-spark-1-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for muse-spark-1-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for muse-spark-1-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for muse-spark-1-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for glm-5-2-none. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for glm-5-2-minimal. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for glm-5-2-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for glm-5-2-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for glm-5-2-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for glm-5-2-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for glm-5-2-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gemini-3-7-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gemini-3-7-flash-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for gemini-3-7-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-sonnet-5-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-sonnet-5-medium. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-sonnet-5-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-sonnet-5-xhigh. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for claude-sonnet-5-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for deepseek-v4-flash-low. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for deepseek-v4-flash-high. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified WebArena-Verified score was ingested for deepseek-v4-flash-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-swe-bench-verified-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "swe-bench-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SWE-bench Verified — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI mini-SWE-agent bash harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-swebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmlu-pro-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "mmlu-pro",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MMLU Pro — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI five-shot MMLU-Pro harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmlu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-briefcase-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "aa-briefcase",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-Briefcase Overall Elo score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-briefcase",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-intelligence-index-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "aa-intelligence-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA Intelligence Index v4.1.1 (score) score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-methodology",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-cursorbench-3-2-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "cursorbench-3-2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CursorBench v3.2 score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-cursorbench-32",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-agents-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "apex-agents",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified APEX-Agents Mean Criteria Passed score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-apex-agents",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "harvey-lab-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey LAB (Vals) score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-harvey-lab",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-jobbench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "jobbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified JobBench score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-babyvision-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "babyvision",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified BabyVision (with CI) score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-perceptionbench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "perceptionbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PerceptionBench score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-verified-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "osworld-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld-Verified score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "apex-swe",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified APEX-SWE (Pass@1, Terminus-2) score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-apex-swe",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mmmu-pro-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "mmmu-pro",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MMMU-Pro score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-mmmu-pro",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-gpqa-diamond-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "gpqa-diamond",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GPQA Diamond score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-aa-gpqa",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-aa-lcr-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "aa-lcr",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AA-LCR score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livebench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "livebench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LiveBench score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-livecodebench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "livecodebench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LiveCodeBench score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-livecodebench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-creative-writing-v3-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "creative-writing-v3",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Creative Writing v3 score was ingested for glm-5-3-max. The requested Kimi K3 row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-paperbench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "paperbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified PaperBench (Replication Score) score was ingested for glm-5-3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-qwenreactbench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "qwenreactbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified QwenReactBench (Elo) score was ingested for glm-5-3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-coworkbench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "coworkbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CoWorkBench score was ingested for glm-5-3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-erqa-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "erqa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ERQA score was ingested for glm-5-3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-lvbench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "lvbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LVBench (with Memory) score was ingested for glm-5-3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vision2web-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "vision2web",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vision2Web (Avg. Frontend/Webpage/etc.) score was ingested for glm-5-3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-mobileworld-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "mobileworld",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MobileWorld score was ingested for glm-5-3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-webarena-verified-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "webarena-verified",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified WebArena-Verified score was ingested for glm-5-3-max. The requested Qwen3.8 Max row is source-matched and other model cells remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "frontiercode",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for FrontierCode v1.1 Extended. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-lab-aa-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "harvey-lab-aa",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for Harvey LAB-AA. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-harvey-lab-aa",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "code-migration",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for Code Migration. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-code-migration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "excel-modeling-benchmark",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for Excel Modeling Benchmark. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "legal-research-bench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for Legal Research Bench. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-vibe-code-bench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "vibe-code-bench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for Vibe Code Bench v1.1. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-vibe-code",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-programbench-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "programbench",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for ProgramBench · Vals run. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-programbench",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-integration-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "apex-swe-integration",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for APEX-SWE · Integration. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-integration",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-apex-swe-observability-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "apex-swe-observability",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified GLM-5.3 score was ingested for APEX-SWE · Observability. The requested GLM-5.3 coverage remains limited to the provider-published rows in the benchmark sheet; no proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-apex-swe-observability",
      "reviewedAt": "2026-08-15T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-opus-5-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-opus-5-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-opus-5-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-opus-5-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-fable-5-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-fable-5-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-fable-5-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-fable-5-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-fable-5-max",
      "modelId": "claude-fable-5-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for claude-fable-5-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-sol-none. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-sol-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-sol-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-sol-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-sol-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-sol-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-terra-none. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-terra-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-terra-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-terra-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-terra-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-terra-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-luna-none. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-luna-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-luna-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-luna-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-luna-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gpt-5-6-luna-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for grok-4-6-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for grok-4-6-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for grok-4-6-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for grok-4-6-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for kimi-k3-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for kimi-k3-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for kimi-k3-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gemini-3-1-pro-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gemini-3-1-pro-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for gemini-3-1-pro-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for deepseek-v4-pro-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for deepseek-v4-pro-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for deepseek-v4-pro-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for qwen-3-8-max-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for qwen-3-8-max-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for qwen-3-8-max-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for muse-spark-1-2-minimal. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for muse-spark-1-2-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for muse-spark-1-2-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for muse-spark-1-2-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for muse-spark-1-2-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for glm-5-2-none. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for glm-5-2-minimal. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for glm-5-2-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for glm-5-2-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for glm-5-2-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for glm-5-2-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for glm-5-2-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for glm-5-3-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gemini-3-7-flash-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for gemini-3-7-flash-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified ExploitBench score was ingested for gemini-3-7-flash-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-sonnet-5-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-sonnet-5-medium. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-sonnet-5-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-sonnet-5-xhigh. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for claude-sonnet-5-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for deepseek-v4-flash-low. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for deepseek-v4-flash-high. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-exploitbench-v8-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "exploitbench-v8",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified ExploitBench score was ingested for deepseek-v4-flash-max. The Opus 5 max result is directly reported in the system card. Other sheet values are retained separately and are not assigned this exact source or configuration. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-opus-5-max",
      "modelId": "claude-opus-5-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-opus-5-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-opus-5-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-opus-5-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-opus-5-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-opus-5-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-fable-5-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-fable-5-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-fable-5-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-fable-5-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-sol-none. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-sol-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-sol-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-sol-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-sol-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-sol-max",
      "modelId": "gpt-5-6-sol-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-sol-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-terra-none. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-terra-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-terra-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-terra-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-terra-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-terra-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-luna-none. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-luna-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-luna-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-luna-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-luna-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gpt-5-6-luna-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for grok-4-6-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for grok-4-6-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for grok-4-6-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for grok-4-6-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for kimi-k3-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for kimi-k3-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gemini-3-1-pro-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gemini-3-1-pro-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gemini-3-1-pro-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for deepseek-v4-pro-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for deepseek-v4-pro-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for deepseek-v4-pro-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for qwen-3-8-max-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for qwen-3-8-max-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for qwen-3-8-max-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for muse-spark-1-2-minimal. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for muse-spark-1-2-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for muse-spark-1-2-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for muse-spark-1-2-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for muse-spark-1-2-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for glm-5-2-none. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for glm-5-2-minimal. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for glm-5-2-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for glm-5-2-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for glm-5-2-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for glm-5-2-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for glm-5-2-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for glm-5-3-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gemini-3-7-flash-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gemini-3-7-flash-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for gemini-3-7-flash-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-sonnet-5-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-sonnet-5-medium. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-sonnet-5-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-sonnet-5-xhigh. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for claude-sonnet-5-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for deepseek-v4-flash-low. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for deepseek-v4-flash-high. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-05-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath May 2026 score was ingested for deepseek-v4-flash-max. Only exact Fable 5 max and Kimi K3 May results are available. GPT-5.6 Sol and Opus 5 have June results and remain blank here. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-opus-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-opus-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-opus-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-opus-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-fable-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-fable-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-fable-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-fable-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-sol-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-sol-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-sol-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-sol-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-sol-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-terra-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-terra-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-terra-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-terra-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-terra-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-terra-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-luna-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-luna-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-luna-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-luna-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-luna-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gpt-5-6-luna-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for grok-4-6-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for grok-4-6-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for grok-4-6-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for grok-4-6-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for kimi-k3-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for kimi-k3-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gemini-3-1-pro-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gemini-3-1-pro-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gemini-3-1-pro-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for deepseek-v4-pro-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for deepseek-v4-pro-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for deepseek-v4-pro-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for qwen-3-8-max-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for qwen-3-8-max-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for qwen-3-8-max-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for muse-spark-1-2-minimal. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for muse-spark-1-2-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for muse-spark-1-2-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for muse-spark-1-2-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for muse-spark-1-2-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for glm-5-2-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for glm-5-2-minimal. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for glm-5-2-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for glm-5-2-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for glm-5-2-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for glm-5-2-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for glm-5-2-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for glm-5-3-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gemini-3-7-flash-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gemini-3-7-flash-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for gemini-3-7-flash-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-sonnet-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-sonnet-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-sonnet-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-sonnet-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for claude-sonnet-5-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for deepseek-v4-flash-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for deepseek-v4-flash-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivmath-2026-06-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivMath Jun 2026 score was ingested for deepseek-v4-flash-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5. It is not merged with the separate May 2026 Fable result. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-opus-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-opus-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-opus-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-opus-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-fable-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-fable-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-fable-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-fable-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-sol-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-sol-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-sol-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-sol-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-sol-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-terra-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-terra-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-terra-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-terra-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-terra-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-terra-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-luna-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-luna-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-luna-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-luna-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-luna-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gpt-5-6-luna-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for grok-4-6-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for grok-4-6-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for grok-4-6-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for grok-4-6-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for kimi-k3-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for kimi-k3-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gemini-3-1-pro-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gemini-3-1-pro-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gemini-3-1-pro-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for deepseek-v4-pro-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for deepseek-v4-pro-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for deepseek-v4-pro-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for qwen-3-8-max-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for qwen-3-8-max-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for qwen-3-8-max-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for muse-spark-1-2-minimal. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for muse-spark-1-2-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for muse-spark-1-2-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for muse-spark-1-2-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for muse-spark-1-2-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for glm-5-2-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for glm-5-2-minimal. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for glm-5-2-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for glm-5-2-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for glm-5-2-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for glm-5-2-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for glm-5-2-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for glm-5-3-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gemini-3-7-flash-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gemini-3-7-flash-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for gemini-3-7-flash-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-sonnet-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-sonnet-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-sonnet-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-sonnet-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for claude-sonnet-5-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for deepseek-v4-flash-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for deepseek-v4-flash-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-brokenarxiv-2026-06-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — BrokenArXiv Jun 2026 score was ingested for deepseek-v4-flash-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-opus-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-opus-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-opus-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-opus-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-fable-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-fable-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-fable-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-fable-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-sol-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-sol-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-sol-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-sol-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-sol-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-terra-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-terra-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-terra-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-terra-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-terra-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-terra-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-luna-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-luna-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-luna-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-luna-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-luna-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gpt-5-6-luna-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for grok-4-6-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for grok-4-6-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for grok-4-6-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for grok-4-6-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for kimi-k3-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for kimi-k3-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for kimi-k3-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gemini-3-1-pro-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gemini-3-1-pro-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gemini-3-1-pro-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for deepseek-v4-pro-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for deepseek-v4-pro-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for deepseek-v4-pro-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for qwen-3-8-max-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for qwen-3-8-max-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for qwen-3-8-max-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for muse-spark-1-2-minimal. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for muse-spark-1-2-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for muse-spark-1-2-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for muse-spark-1-2-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for muse-spark-1-2-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for glm-5-2-none. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for glm-5-2-minimal. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for glm-5-2-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for glm-5-2-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for glm-5-2-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for glm-5-2-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for glm-5-2-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for glm-5-3-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gemini-3-7-flash-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gemini-3-7-flash-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for gemini-3-7-flash-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-sonnet-5-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-sonnet-5-medium. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-sonnet-5-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-sonnet-5-xhigh. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for claude-sonnet-5-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for deepseek-v4-flash-low. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for deepseek-v4-flash-high. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-matharena-arxivlean-2026-06-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MathArena — ArXivLean Jun 2026 score was ingested for deepseek-v4-flash-max. The June competition has exact max-configuration results for Fable 5, GPT-5.6 Sol, and Opus 5; other model cells remain missing. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-matharena",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-opus-5-xhigh. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-opus-5-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-opus-5-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-opus-5-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-fable-5-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-fable-5-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-fable-5-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-fable-5-xhigh. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-sol-none. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-sol-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-sol-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-sol-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-sol-xhigh. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-terra-none. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-terra-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-terra-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-terra-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-terra-xhigh. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-terra-max. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-luna-none. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-luna-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-luna-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-luna-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-luna-xhigh. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gpt-5-6-luna-max. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for grok-4-6-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for grok-4-6-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for grok-4-6-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for kimi-k3-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for kimi-k3-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gemini-3-1-pro-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gemini-3-1-pro-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SciCode score was ingested for gemini-3-1-pro-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for deepseek-v4-pro-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for deepseek-v4-pro-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for qwen-3-8-max-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for qwen-3-8-max-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for muse-spark-1-2-minimal. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for muse-spark-1-2-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for muse-spark-1-2-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for muse-spark-1-2-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for glm-5-2-none. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for glm-5-2-minimal. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for glm-5-2-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for glm-5-2-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for glm-5-2-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for glm-5-2-xhigh. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for glm-5-2-max. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SciCode score was ingested for glm-5-3-max. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gemini-3-7-flash-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for gemini-3-7-flash-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-sonnet-5-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-sonnet-5-medium. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-sonnet-5-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-sonnet-5-xhigh. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for claude-sonnet-5-max. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for deepseek-v4-flash-low. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for deepseek-v4-flash-high. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-scicode-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "scicode",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SciCode score was ingested for deepseek-v4-flash-max. Directly reported frontier values are retained. Other cells stay blank and no score is copied from the older maintainer leaderboard. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-sci-code",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-opus-5-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-opus-5-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-opus-5-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-opus-5-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-fable-5-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-fable-5-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-fable-5-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-fable-5-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-sol-none. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-sol-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-sol-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-sol-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-sol-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-terra-none. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-terra-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-terra-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-terra-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-terra-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-terra-max. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-luna-none. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-luna-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-luna-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-luna-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-luna-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gpt-5-6-luna-max. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for grok-4-6-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for grok-4-6-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for grok-4-6-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for kimi-k3-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for kimi-k3-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gemini-3-1-pro-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gemini-3-1-pro-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gemini-3-1-pro-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for deepseek-v4-pro-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for deepseek-v4-pro-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for qwen-3-8-max-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for qwen-3-8-max-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for qwen-3-8-max-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for muse-spark-1-2-minimal. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for muse-spark-1-2-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for muse-spark-1-2-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for muse-spark-1-2-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for muse-spark-1-2-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for glm-5-2-none. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for glm-5-2-minimal. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for glm-5-2-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for glm-5-2-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for glm-5-2-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for glm-5-2-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for glm-5-2-max. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for glm-5-3-max. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gemini-3-7-flash-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for gemini-3-7-flash-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-sonnet-5-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-sonnet-5-medium. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-sonnet-5-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-sonnet-5-xhigh. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for claude-sonnet-5-max. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for deepseek-v4-flash-low. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for deepseek-v4-flash-high. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontiercode-1-1-main-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "frontiercode-1-1-main",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified FrontierCode v1.1 Main score was ingested for deepseek-v4-flash-max. Exact Main-subset results are retained for the seven populated models. Missing models are not filled from Extended results or older FrontierCode versions. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-frontiercode-11",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-opus-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-opus-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-opus-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-opus-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-fable-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-fable-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-fable-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-fable-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-sol-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-sol-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-sol-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-sol-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-sol-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-terra-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-terra-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-terra-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-terra-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-terra-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-terra-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-luna-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-luna-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-luna-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-luna-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-luna-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gpt-5-6-luna-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for grok-4-6-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for grok-4-6-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for grok-4-6-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for kimi-k3-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for kimi-k3-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gemini-3-1-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gemini-3-1-pro-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gemini-3-1-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for deepseek-v4-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for deepseek-v4-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for qwen-3-8-max-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for qwen-3-8-max-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for muse-spark-1-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for muse-spark-1-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for muse-spark-1-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for muse-spark-1-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for glm-5-2-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for glm-5-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for glm-5-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for glm-5-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for glm-5-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for glm-5-2-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for glm-5-2-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gemini-3-7-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for gemini-3-7-flash-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-sonnet-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-sonnet-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-sonnet-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-sonnet-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for claude-sonnet-5-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for deepseek-v4-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for deepseek-v4-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-terminal-bench-2-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "terminal-bench-2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Terminal-Bench 2.1 score was ingested for deepseek-v4-flash-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-terminal-bench",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SAGE — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SAGE — Vals score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SAGE — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SAGE — Vals score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-sage-vals-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "sage-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SAGE — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for grok-4-6-xhigh. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-corpfin-v2-vals-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "corpfin-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified CorpFin v2 — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI CorpFin v2 evaluation with Sonnet 4.5 judge; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-corpfin-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MortgageTax — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MortgageTax — Vals score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MortgageTax — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MortgageTax — Vals score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mortgage-tax-vals-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "mortgage-tax-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MortgageTax — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-finance-agent-v2-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "finance-agent-v2",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Finance Agent v2 — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI Finance Agent v2 harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-finance-agent-v2",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-excel-modeling-benchmark-vals-overall-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Excel Modeling Benchmark — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI Excel Modeling overall evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-excel-modeling",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-taxeval-v2-vals-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "taxeval-v2-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified TaxEval v2 — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI max-compute evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MedCode — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MedCode — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MedCode — Vals score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medcode-vals-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "medcode-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedCode — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MedScribe — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MedScribe — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MedScribe — Vals score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-medscribe-vals-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "medscribe-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MedScribe — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI health evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vals Index score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vals Index score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-index-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "vals-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Index score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vals Multimodal Index score was ingested for grok-4-6-xhigh. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vals Multimodal Index score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vals Multimodal Index score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vals Multimodal Index score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Vals Multimodal Index score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-vals-multimodal-index-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "vals-multimodal-index",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Vals Multimodal Index score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI multimodal index methodology; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legal-research-bench-vals-overall-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "legal-research-bench-vals-overall",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Legal Research Bench — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI Legal Research overall harness; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-vals-legal-research",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LegalBench — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LegalBench — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified LegalBench — Vals score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-legalbench-vals-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "legalbench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified LegalBench — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI LegalBench evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for grok-4-6-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for grok-4-6-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for kimi-k3-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for kimi-k3-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for glm-5-2-none. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for glm-5-2-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for glm-5-2-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for glm-5-2-max. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for glm-5-3-max. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-public-benefits-bench-vals-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "public-benefits-bench-vals",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Public Benefits Bench — Vals score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Vals AI public-benefits evaluation; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-opus-5-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-opus-5-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-fable-5-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-fable-5-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for grok-4-6-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for grok-4-6-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for grok-4-6-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AIIQ Composite IQ score was ingested for grok-4-6-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for kimi-k3-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for kimi-k3-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AIIQ Composite IQ score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AIIQ Composite IQ score was ingested for qwen-3-8-max-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AIIQ Composite IQ score was ingested for muse-spark-1-2-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for glm-5-2-none. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for glm-5-2-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for glm-5-2-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for glm-5-2-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for glm-5-2-max. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AIIQ Composite IQ score was ingested for glm-5-3-max. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AIIQ Composite IQ score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-aiiq-composite-iq-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "aiiq-composite-iq",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AIIQ Composite IQ score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under AIIQ composite snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for grok-4-6-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for grok-4-6-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Design Arena (Elo) score was ingested for grok-4-6-xhigh. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for kimi-k3-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for kimi-k3-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Design Arena (Elo) score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Design Arena (Elo) score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for glm-5-2-none. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for glm-5-2-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for glm-5-2-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for glm-5-2-max. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Design Arena (Elo) score was ingested for glm-5-3-max. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Design Arena (Elo) score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-design-arena-elo-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "design-arena-elo",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Design Arena (Elo) score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Design Arena 2026-08-11 rating snapshot; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for grok-4-6-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for grok-4-6-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for grok-4-6-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for kimi-k3-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for kimi-k3-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for kimi-k3-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for qwen-3-8-max-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for muse-spark-1-2-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for glm-5-2-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for glm-5-2-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for glm-5-2-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for glm-5-2-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for glm-5-3-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-frontier-bench-v0-1-anthropic-h2h-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Frontier-Bench v0.1 — Anthropic H2H score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for grok-4-6-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for grok-4-6-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for grok-4-6-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for kimi-k3-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for kimi-k3-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for muse-spark-1-2-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for glm-5-2-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for glm-5-2-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for glm-5-2-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for glm-5-2-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for glm-5-3-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-osworld-2-anthropic-h2h-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OSWorld 2.0 — Anthropic H2H score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-opus-5-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-opus-5-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-opus-5-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-opus-5-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-fable-5-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-fable-5-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-fable-5-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-fable-5-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-sol-none. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-sol-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-sol-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-sol-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-sol-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-terra-none. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-terra-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-terra-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-terra-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-terra-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-terra-max. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-luna-none. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-luna-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-luna-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-luna-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-luna-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gpt-5-6-luna-max. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for grok-4-6-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for grok-4-6-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for grok-4-6-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for grok-4-6-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for kimi-k3-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for kimi-k3-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gemini-3-1-pro-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gemini-3-1-pro-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gemini-3-1-pro-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for deepseek-v4-pro-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for deepseek-v4-pro-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for qwen-3-8-max-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for qwen-3-8-max-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for muse-spark-1-2-minimal. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for muse-spark-1-2-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for muse-spark-1-2-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for muse-spark-1-2-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for muse-spark-1-2-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for glm-5-2-none. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for glm-5-2-minimal. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for glm-5-2-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for glm-5-2-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for glm-5-2-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for glm-5-2-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for glm-5-2-max. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gemini-3-7-flash-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for gemini-3-7-flash-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-sonnet-5-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-sonnet-5-medium. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-sonnet-5-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-sonnet-5-xhigh. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for claude-sonnet-5-max. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for deepseek-v4-flash-low. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for deepseek-v4-flash-high. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-v1-0-6-public-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "automationbench-v1-0-6-public",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench v1.0.6 — public score was ingested for deepseek-v4-flash-max. Public-set results only; the private 26.0% Opus result remains in the separate held-out comparison row. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-automationbench-public",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for grok-4-6-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for grok-4-6-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for grok-4-6-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for kimi-k3-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for kimi-k3-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for kimi-k3-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for qwen-3-8-max-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for muse-spark-1-2-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for glm-5-2-none. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for glm-5-2-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for glm-5-2-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for glm-5-2-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for glm-5-3-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-automationbench-anthropic-h2h-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "automationbench-anthropic-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified AutomationBench — Anthropic H2H score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Anthropic Opus 5 published head-to-head setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-opus-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-opus-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-opus-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-opus-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-fable-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-fable-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-fable-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-fable-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-sol-none. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-sol-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-sol-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-sol-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-sol-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-terra-none. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-terra-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-terra-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-terra-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-terra-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-terra-max. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-luna-none. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-luna-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-luna-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-luna-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-luna-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gpt-5-6-luna-max. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for grok-4-6-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for grok-4-6-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for grok-4-6-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for grok-4-6-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for kimi-k3-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for kimi-k3-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for kimi-k3-max. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gemini-3-1-pro-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gemini-3-1-pro-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gemini-3-1-pro-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for deepseek-v4-pro-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for deepseek-v4-pro-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for deepseek-v4-pro-max. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for qwen-3-8-max-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for qwen-3-8-max-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for qwen-3-8-max-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for muse-spark-1-2-minimal. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for muse-spark-1-2-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for muse-spark-1-2-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for muse-spark-1-2-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for muse-spark-1-2-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for glm-5-2-none. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for glm-5-2-minimal. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for glm-5-2-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for glm-5-2-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for glm-5-2-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for glm-5-2-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for glm-5-2-max. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for glm-5-3-max. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gemini-3-7-flash-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gemini-3-7-flash-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for gemini-3-7-flash-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-sonnet-5-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-sonnet-5-medium. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-sonnet-5-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-sonnet-5-xhigh. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for claude-sonnet-5-max. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for deepseek-v4-flash-low. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for deepseek-v4-flash-high. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-harvey-legal-agent-held-out-h2h-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Harvey Legal Agent Benchmark — held-out score was ingested for deepseek-v4-flash-max. Requested max-configuration values are retained under Anthropic Opus 5 published held-out legal-agent setup; absent models remain blank. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-opus-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-opus-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-opus-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-opus-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-fable-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-fable-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-fable-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-fable-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-sol-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-sol-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-sol-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-sol-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-sol-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-terra-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-terra-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-terra-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-terra-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-terra-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-terra-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-luna-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-luna-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-luna-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-luna-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-luna-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gpt-5-6-luna-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for grok-4-6-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for grok-4-6-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for grok-4-6-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MCP Atlas † score was ingested for grok-4-6-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for kimi-k3-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for kimi-k3-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gemini-3-1-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gemini-3-1-pro-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MCP Atlas † score was ingested for gemini-3-1-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for deepseek-v4-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for deepseek-v4-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for qwen-3-8-max-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for qwen-3-8-max-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MCP Atlas † score was ingested for qwen-3-8-max-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for muse-spark-1-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for muse-spark-1-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for muse-spark-1-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for muse-spark-1-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for glm-5-2-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for glm-5-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for glm-5-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for glm-5-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for glm-5-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for glm-5-2-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for glm-5-2-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MCP Atlas † score was ingested for glm-5-3-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gemini-3-7-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for gemini-3-7-flash-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified MCP Atlas † score was ingested for gemini-3-7-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-sonnet-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-sonnet-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-sonnet-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-sonnet-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for claude-sonnet-5-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for deepseek-v4-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for deepseek-v4-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-mcp-atlas-cross-source-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "mcp-atlas-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified MCP Atlas † score was ingested for deepseek-v4-flash-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-opus-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-opus-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-opus-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-opus-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-fable-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-fable-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-fable-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-fable-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-sol-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-sol-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-sol-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-sol-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-sol-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-terra-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-terra-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-terra-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-terra-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-terra-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-terra-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-luna-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-luna-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-luna-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-luna-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-luna-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gpt-5-6-luna-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for grok-4-6-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for grok-4-6-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for grok-4-6-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for kimi-k3-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for kimi-k3-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gemini-3-1-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gemini-3-1-pro-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Code Migration † score was ingested for gemini-3-1-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for deepseek-v4-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for deepseek-v4-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for qwen-3-8-max-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for qwen-3-8-max-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Code Migration † score was ingested for qwen-3-8-max-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for muse-spark-1-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for muse-spark-1-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for muse-spark-1-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for muse-spark-1-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for glm-5-2-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for glm-5-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for glm-5-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for glm-5-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for glm-5-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for glm-5-2-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for glm-5-2-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Code Migration † score was ingested for glm-5-3-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gemini-3-7-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for gemini-3-7-flash-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified Code Migration † score was ingested for gemini-3-7-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-sonnet-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-sonnet-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-sonnet-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-sonnet-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for claude-sonnet-5-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for deepseek-v4-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for deepseek-v4-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-code-migration-cross-source-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "code-migration-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified Code Migration † score was ingested for deepseek-v4-flash-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-opus-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-opus-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-opus-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-opus-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-fable-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-fable-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-fable-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-fable-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-sol-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-sol-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-sol-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-sol-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-sol-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-terra-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-terra-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-terra-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-terra-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-terra-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-terra-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-luna-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-luna-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-luna-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-luna-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-luna-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gpt-5-6-luna-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for grok-4-6-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for grok-4-6-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for grok-4-6-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-grok-4-6-xhigh",
      "modelId": "grok-4-6-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SkillsBench † score was ingested for grok-4-6-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for kimi-k3-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for kimi-k3-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-kimi-k3-max",
      "modelId": "kimi-k3-max",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SkillsBench † score was ingested for kimi-k3-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gemini-3-1-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gemini-3-1-pro-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SkillsBench † score was ingested for gemini-3-1-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for deepseek-v4-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for deepseek-v4-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SkillsBench † score was ingested for deepseek-v4-pro-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for qwen-3-8-max-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for qwen-3-8-max-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for muse-spark-1-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for muse-spark-1-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for muse-spark-1-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for muse-spark-1-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for glm-5-2-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for glm-5-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for glm-5-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for glm-5-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for glm-5-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for glm-5-2-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for glm-5-2-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SkillsBench † score was ingested for glm-5-3-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gemini-3-7-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for gemini-3-7-flash-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified SkillsBench † score was ingested for gemini-3-7-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-sonnet-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-sonnet-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-sonnet-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-sonnet-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for claude-sonnet-5-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for deepseek-v4-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for deepseek-v4-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-skillsbench-cross-source-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "skillsbench-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified SkillsBench † score was ingested for deepseek-v4-flash-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-opus-5-xhigh",
      "modelId": "claude-opus-5-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-opus-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-opus-5-high",
      "modelId": "claude-opus-5-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-opus-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-opus-5-medium",
      "modelId": "claude-opus-5-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-opus-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-opus-5-low",
      "modelId": "claude-opus-5-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-opus-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-fable-5-low",
      "modelId": "claude-fable-5-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-fable-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-fable-5-medium",
      "modelId": "claude-fable-5-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-fable-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-fable-5-high",
      "modelId": "claude-fable-5-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-fable-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-fable-5-xhigh",
      "modelId": "claude-fable-5-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-fable-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-sol-none",
      "modelId": "gpt-5-6-sol-none",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-sol-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-sol-low",
      "modelId": "gpt-5-6-sol-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-sol-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-sol-medium",
      "modelId": "gpt-5-6-sol-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-sol-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-sol-high",
      "modelId": "gpt-5-6-sol-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-sol-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-sol-xhigh",
      "modelId": "gpt-5-6-sol-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-sol-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-terra-none",
      "modelId": "gpt-5-6-terra-none",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-terra-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-terra-low",
      "modelId": "gpt-5-6-terra-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-terra-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-terra-medium",
      "modelId": "gpt-5-6-terra-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-terra-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-terra-high",
      "modelId": "gpt-5-6-terra-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-terra-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-terra-xhigh",
      "modelId": "gpt-5-6-terra-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-terra-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-terra-max",
      "modelId": "gpt-5-6-terra-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-terra-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-luna-none",
      "modelId": "gpt-5-6-luna-none",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-luna-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-luna-low",
      "modelId": "gpt-5-6-luna-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-luna-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-luna-medium",
      "modelId": "gpt-5-6-luna-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-luna-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-luna-high",
      "modelId": "gpt-5-6-luna-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-luna-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-luna-xhigh",
      "modelId": "gpt-5-6-luna-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-luna-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gpt-5-6-luna-max",
      "modelId": "gpt-5-6-luna-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gpt-5-6-luna-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-grok-4-6-low",
      "modelId": "grok-4-6-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for grok-4-6-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-grok-4-6-medium",
      "modelId": "grok-4-6-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for grok-4-6-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-grok-4-6-high",
      "modelId": "grok-4-6-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for grok-4-6-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-kimi-k3-low",
      "modelId": "kimi-k3-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for kimi-k3-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-kimi-k3-high",
      "modelId": "kimi-k3-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for kimi-k3-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gemini-3-1-pro-low",
      "modelId": "gemini-3-1-pro-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gemini-3-1-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gemini-3-1-pro-medium",
      "modelId": "gemini-3-1-pro-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gemini-3-1-pro-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gemini-3-1-pro-high",
      "modelId": "gemini-3-1-pro-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OfficeQA Pro † score was ingested for gemini-3-1-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-deepseek-v4-pro-low",
      "modelId": "deepseek-v4-pro-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for deepseek-v4-pro-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-deepseek-v4-pro-high",
      "modelId": "deepseek-v4-pro-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for deepseek-v4-pro-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-deepseek-v4-pro-max",
      "modelId": "deepseek-v4-pro-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OfficeQA Pro † score was ingested for deepseek-v4-pro-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-qwen-3-8-max-low",
      "modelId": "qwen-3-8-max-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for qwen-3-8-max-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-qwen-3-8-max-medium",
      "modelId": "qwen-3-8-max-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for qwen-3-8-max-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-qwen-3-8-max-xhigh",
      "modelId": "qwen-3-8-max-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OfficeQA Pro † score was ingested for qwen-3-8-max-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-muse-spark-1-2-minimal",
      "modelId": "muse-spark-1-2-minimal",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for muse-spark-1-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-muse-spark-1-2-low",
      "modelId": "muse-spark-1-2-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for muse-spark-1-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-muse-spark-1-2-medium",
      "modelId": "muse-spark-1-2-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for muse-spark-1-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-muse-spark-1-2-high",
      "modelId": "muse-spark-1-2-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for muse-spark-1-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-muse-spark-1-2-xhigh",
      "modelId": "muse-spark-1-2-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OfficeQA Pro † score was ingested for muse-spark-1-2-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-glm-5-2-none",
      "modelId": "glm-5-2-none",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for glm-5-2-none. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-glm-5-2-minimal",
      "modelId": "glm-5-2-minimal",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for glm-5-2-minimal. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-glm-5-2-low",
      "modelId": "glm-5-2-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for glm-5-2-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-glm-5-2-medium",
      "modelId": "glm-5-2-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for glm-5-2-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-glm-5-2-high",
      "modelId": "glm-5-2-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for glm-5-2-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-glm-5-2-xhigh",
      "modelId": "glm-5-2-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for glm-5-2-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-glm-5-2-max",
      "modelId": "glm-5-2-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for glm-5-2-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-glm-5-3-max",
      "modelId": "glm-5-3-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OfficeQA Pro † score was ingested for glm-5-3-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gemini-3-7-flash-low",
      "modelId": "gemini-3-7-flash-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gemini-3-7-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gemini-3-7-flash-medium",
      "modelId": "gemini-3-7-flash-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for gemini-3-7-flash-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-gemini-3-7-flash-high",
      "modelId": "gemini-3-7-flash-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "not_evaluated",
      "note": "No directly verified OfficeQA Pro † score was ingested for gemini-3-7-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-sonnet-5-low",
      "modelId": "claude-sonnet-5-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-sonnet-5-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-sonnet-5-medium",
      "modelId": "claude-sonnet-5-medium",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-sonnet-5-medium. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-sonnet-5-high",
      "modelId": "claude-sonnet-5-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-sonnet-5-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-sonnet-5-xhigh",
      "modelId": "claude-sonnet-5-xhigh",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-sonnet-5-xhigh. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-claude-sonnet-5-max",
      "modelId": "claude-sonnet-5-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for claude-sonnet-5-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-deepseek-v4-flash-low",
      "modelId": "deepseek-v4-flash-low",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for deepseek-v4-flash-low. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-deepseek-v4-flash-high",
      "modelId": "deepseek-v4-flash-high",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for deepseek-v4-flash-high. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    },
    {
      "id": "coverage-officeqa-pro-cross-source-deepseek-v4-flash-max",
      "modelId": "deepseek-v4-flash-max",
      "benchmarkId": "officeqa-pro-cross-source",
      "status": "missing",
      "missingReason": "result_unavailable",
      "note": "No directly verified OfficeQA Pro † score was ingested for deepseek-v4-flash-max. Dagger-marked comparison: useful directional evidence, but public harness provenance is not fully controlled. No proxy, fallback, or model-family substitution is inferred.",
      "citationId": "citation-benchmarklist-fable-5",
      "reviewedAt": "2026-08-16T00:00:00.000Z"
    }
  ],
  "revisions": [
    {
      "id": "revision-remove-fixture-rows",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "reclassified",
      "field": "legacy fixture benchmark rows",
      "previousValue": "six conceptual rows with illustrative values",
      "newValue": "rejected audit records; no production result rows",
      "reason": "A category may be conceptual, but a benchmark row must identify a real public evaluation, native metric, and public result source.",
      "actor": "benchmark-editorial"
    },
    {
      "id": "revision-verified-suite-freeze",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "field": "verified benchmark suite",
      "newValue": "gdpval-aa-v2, gpqa-diamond",
      "reason": "Only rows with a pinned definition, native metric, public source, and structured requested-roster observations enter the current artifact.",
      "actor": "benchmark-editorial"
    },
    {
      "id": "revision-freeze-score-calibration",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "field": "latent score calibration",
      "newValue": "median center + IQR/1.349 scale over comparable native cohort",
      "reason": "The preview score's calibration parameters are derived from the verified GDPval-AA v2 and GPQA Diamond cohort and are included in the downloadable score config.",
      "actor": "benchmark-editorial"
    },
    {
      "id": "revision-admit-frontier-threshold",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "reclassified",
      "field": "benchmark admission policy",
      "previousValue": "verified rows could score despite unresolved target coverage",
      "newValue": "included rows require at most two missing frontier target cells; caveats are surfaced separately",
      "reason": "The resilient index must not silently publish a benchmark with materially incomplete frontier evidence.",
      "actor": "benchmark-governance-v2"
    },
    {
      "id": "revision-ingest-code-migration",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "benchmarkId": "code-migration",
      "field": "benchmark definition and public observations",
      "newValue": "Vals Code Migration · CLI split · 2026-08-15 snapshot; 7 directly readable observations",
      "reason": "Added the requested research candidate with its native metric, source snapshot, explicit missing coverage, and admission exclusion. Unverified effort settings and absent rows remain visible instead of being imputed.",
      "citationId": "citation-vals-code-migration",
      "actor": "benchmark-ingestion-v3"
    },
    {
      "id": "revision-ingest-excel-modeling-benchmark",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "benchmarkId": "excel-modeling-benchmark",
      "field": "benchmark definition and public observations",
      "newValue": "Vals Excel Modeling Benchmark · Scratch Dataroom slice · 2026-08-15 snapshot; 7 directly readable observations",
      "reason": "Added the requested research candidate with its native metric, source snapshot, explicit missing coverage, and admission exclusion. Unverified effort settings and absent rows remain visible instead of being imputed.",
      "citationId": "citation-vals-excel-modeling",
      "actor": "benchmark-ingestion-v3"
    },
    {
      "id": "revision-ingest-legal-research-bench",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "benchmarkId": "legal-research-bench",
      "field": "benchmark definition and public observations",
      "newValue": "Vals Legal Research Bench · Health-area slice · 2026-08-15 snapshot; 7 directly readable observations",
      "reason": "Added the requested research candidate with its native metric, source snapshot, explicit missing coverage, and admission exclusion. Unverified effort settings and absent rows remain visible instead of being imputed.",
      "citationId": "citation-vals-legal-research",
      "actor": "benchmark-ingestion-v3"
    },
    {
      "id": "revision-ingest-vibe-code-bench",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "benchmarkId": "vibe-code-bench",
      "field": "benchmark definition and public observations",
      "newValue": "Vals Vibe Code Bench v1.1 · public leaderboard snapshot · 2026-08-15; 2 directly readable observations",
      "reason": "Added the requested research candidate with its native metric, source snapshot, explicit missing coverage, and admission exclusion. Unverified effort settings and absent rows remain visible instead of being imputed.",
      "citationId": "citation-vals-vibe-code",
      "actor": "benchmark-ingestion-v3"
    },
    {
      "id": "revision-ingest-programbench",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "benchmarkId": "programbench",
      "field": "benchmark definition and public observations",
      "newValue": "Vals ProgramBench · public leaderboard snapshot · 2026-08-15; 4 directly readable observations",
      "reason": "Added the requested research candidate with its native metric, source snapshot, explicit missing coverage, and admission exclusion. Unverified effort settings and absent rows remain visible instead of being imputed.",
      "citationId": "citation-vals-programbench",
      "actor": "benchmark-ingestion-v3"
    },
    {
      "id": "revision-ingest-apex-swe-integration",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "benchmarkId": "apex-swe-integration",
      "field": "track-specific benchmark definition and observations",
      "newValue": "APEX-SWE Integration track · public leaderboard snapshot; 5 exact target observations",
      "reason": "Added the APEX-SWE track as a separate benchmark because the aggregate leaderboard fails the frontier gate while both public tracks reach five exact requested configurations. The Terminus-2 versus Archipelago scaffold discrepancy remains an explicit exclusion.",
      "citationId": "citation-apex-swe-integration",
      "actor": "benchmark-ingestion-v3"
    },
    {
      "id": "revision-ingest-apex-swe-observability",
      "revisedAt": "2026-08-15T00:00:00.000Z",
      "changeType": "added",
      "benchmarkId": "apex-swe-observability",
      "field": "track-specific benchmark definition and observations",
      "newValue": "APEX-SWE Observability track · public leaderboard snapshot; 5 exact target observations",
      "reason": "Added the APEX-SWE track as a separate benchmark because the aggregate leaderboard fails the frontier gate while both public tracks reach five exact requested configurations. The Terminus-2 versus Archipelago scaffold discrepancy remains an explicit exclusion.",
      "citationId": "citation-apex-swe-observability",
      "actor": "benchmark-ingestion-v3"
    },
    {
      "id": "revision-ingest-sheet-swe-bench-verified",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "swe-bench-verified",
      "field": "benchmark definition and requested observations",
      "newValue": "Vals SWE-bench Verified · 500-task mini-SWE-agent snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-vals-swebench",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-exploitbench-v8",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "exploitbench-v8",
      "field": "benchmark definition and requested observations",
      "newValue": "ExploitBench V8 · AutoNudge arm · 2026-07-24; 1 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-anthropic-opus-5-system-card",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-matharena-arxivmath-2026-05",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "matharena-arxivmath-2026-05",
      "field": "benchmark definition and requested observations",
      "newValue": "MathArena ArXivMath · 05/2026; 2 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-matharena",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-matharena-arxivmath-2026-06",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "matharena-arxivmath-2026-06",
      "field": "benchmark definition and requested observations",
      "newValue": "MathArena ArXivMath · 06/2026; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-matharena",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-matharena-brokenarxiv-2026-06",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "matharena-brokenarxiv-2026-06",
      "field": "benchmark definition and requested observations",
      "newValue": "MathArena BrokenArXiv · 06/2026; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-matharena",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-matharena-arxivlean-2026-06",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "matharena-arxivlean-2026-06",
      "field": "benchmark definition and requested observations",
      "newValue": "MathArena ArXivLean · 06/2026; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-matharena",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-scicode",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "scicode",
      "field": "benchmark definition and requested observations",
      "newValue": "SciCode · Together AI comparison snapshot; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-sci-code",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-frontiercode",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "frontiercode",
      "field": "benchmark definition and requested observations",
      "newValue": "FrontierCode v1.1 Extended; 5 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-frontiercode-11-data",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-frontiercode-1-1-main",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "frontiercode-1-1-main",
      "field": "benchmark definition and requested observations",
      "newValue": "FrontierCode v1.1 Main · public leaderboard snapshot · 2026-08-16; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-frontiercode-11",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-terminal-bench-2",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "terminal-bench-2",
      "field": "benchmark definition and requested observations",
      "newValue": "Terminal-Bench v2.1 · cross-source snapshot; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-terminal-bench",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-sage-vals",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "sage-vals",
      "field": "benchmark definition and requested observations",
      "newValue": "SAGE · Vals snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-corpfin-v2-vals",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "corpfin-v2-vals",
      "field": "benchmark definition and requested observations",
      "newValue": "CorpFin v2 · Vals archived snapshot; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-vals-corpfin-v2",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-mortgage-tax-vals",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "mortgage-tax-vals",
      "field": "benchmark definition and requested observations",
      "newValue": "MortgageTax · Vals snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-finance-agent-v2",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "finance-agent-v2",
      "field": "benchmark definition and requested observations",
      "newValue": "Vals Finance Agent v2 · public leaderboard snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-vals-finance-agent-v2",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-excel-modeling-benchmark-vals-overall",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "excel-modeling-benchmark-vals-overall",
      "field": "benchmark definition and requested observations",
      "newValue": "Vals Excel Modeling Benchmark · overall snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-vals-excel-modeling",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-taxeval-v2-vals",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "taxeval-v2-vals",
      "field": "benchmark definition and requested observations",
      "newValue": "TaxEval v2 · Vals snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-medcode-vals",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "medcode-vals",
      "field": "benchmark definition and requested observations",
      "newValue": "MedCode · Vals snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-medscribe-vals",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "medscribe-vals",
      "field": "benchmark definition and requested observations",
      "newValue": "MedScribe · Vals snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-mmlu-pro",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "mmlu-pro",
      "field": "benchmark definition and requested observations",
      "newValue": "Vals MMLU-Pro · five-shot public snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-vals-mmlu-pro",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-vals-index",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "vals-index",
      "field": "benchmark definition and requested observations",
      "newValue": "Vals Index snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-vals-multimodal-index",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "vals-multimodal-index",
      "field": "benchmark definition and requested observations",
      "newValue": "Vals Multimodal Index snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-legal-research-bench-vals-overall",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "legal-research-bench-vals-overall",
      "field": "benchmark definition and requested observations",
      "newValue": "Vals Legal Research Bench · overall snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-vals-legal-research",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-legalbench-vals",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "legalbench-vals",
      "field": "benchmark definition and requested observations",
      "newValue": "LegalBench · Vals snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-public-benefits-bench-vals",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "public-benefits-bench-vals",
      "field": "benchmark definition and requested observations",
      "newValue": "Public Benefits Bench · Vals snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aiiq-composite-iq",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aiiq-composite-iq",
      "field": "benchmark definition and requested observations",
      "newValue": "AIIQ Composite IQ snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-design-arena-elo",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "design-arena-elo",
      "field": "benchmark definition and requested observations",
      "newValue": "Design Arena rating snapshot · 2026-08-11; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-frontier-bench-v0-1-anthropic-h2h",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "frontier-bench-v0-1-anthropic-h2h",
      "field": "benchmark definition and requested observations",
      "newValue": "Frontier-Bench v0.1 · Anthropic Opus 5 head-to-head; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-osworld-2-anthropic-h2h",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "osworld-2-anthropic-h2h",
      "field": "benchmark definition and requested observations",
      "newValue": "OSWorld 2.0 · Anthropic Opus 5 head-to-head; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-osworld-verified",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "osworld-verified",
      "field": "benchmark definition and requested observations",
      "newValue": "OSWorld-Verified · cross-source snapshot · 2026-08-16; 5 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-ouroboros-osworld-opus-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-automationbench-v1-0-6-public",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "automationbench-v1-0-6-public",
      "field": "benchmark definition and requested observations",
      "newValue": "AutomationBench public v1.0.6 · 600-task set · 2026-08-16; 8 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-automationbench-public",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-automationbench-anthropic-h2h",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "automationbench-anthropic-h2h",
      "field": "benchmark definition and requested observations",
      "newValue": "AutomationBench · Anthropic Opus 5 head-to-head; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-harvey-legal-agent-held-out-h2h",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "harvey-legal-agent-held-out-h2h",
      "field": "benchmark definition and requested observations",
      "newValue": "Legal Agent Benchmark held-out · Anthropic Opus 5 head-to-head; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-tosea-opus-5-head-to-head",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-mcp-atlas-cross-source",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "mcp-atlas-cross-source",
      "field": "benchmark definition and requested observations",
      "newValue": "MCP Atlas · cross-source snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-code-migration-cross-source",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "code-migration-cross-source",
      "field": "benchmark definition and requested observations",
      "newValue": "Code Migration · cross-source snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-skillsbench-cross-source",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "skillsbench-cross-source",
      "field": "benchmark definition and requested observations",
      "newValue": "SkillsBench · cross-source snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-officeqa-pro-cross-source",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "officeqa-pro-cross-source",
      "field": "benchmark definition and requested observations",
      "newValue": "OfficeQA Pro · cross-source snapshot · 2026-08-15; 10 observations",
      "reason": "Imported from the requested benchmark sheet with source and harness metadata. Existing blank cells remain missing, and distinct months, subsets, versions, and harnesses are not merged.",
      "citationId": "citation-benchmarklist-fable-5",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gdpval-aa-v2-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gdpval-aa-v2",
      "field": "Kimi K3 Max observation",
      "newValue": "1682",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-briefcase-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-briefcase",
      "field": "Kimi K3 Max observation",
      "newValue": "1541",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-intelligence-index-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-intelligence-index",
      "field": "Kimi K3 Max observation",
      "newValue": "60",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-frontiermath-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "frontiermath",
      "field": "Kimi K3 Max observation",
      "newValue": "39.02%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-deepswe-1-1-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "deepswe-1-1",
      "field": "Kimi K3 Max observation",
      "newValue": "69%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-cursorbench-3-2-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "cursorbench-3-2",
      "field": "Kimi K3 Max observation",
      "newValue": "60.8%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-apex-agents-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "apex-agents",
      "field": "Kimi K3 Max observation",
      "newValue": "55.4%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-terminal-bench-3-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "terminal-bench-3",
      "field": "Kimi K3 Max observation",
      "newValue": "17.4%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-explainx-terminal-bench-3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-agents-last-exam-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "agents-last-exam",
      "field": "Kimi K3 Max observation",
      "newValue": "28.3%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-snorkel-agents-last-exam",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-harvey-lab-vals-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "harvey-lab-vals",
      "field": "Kimi K3 Max observation",
      "newValue": "10.83%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-frontierswe-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "frontierswe",
      "field": "Kimi K3 Max observation",
      "newValue": "81.2%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-jobbench-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "jobbench",
      "field": "Kimi K3 Max observation",
      "newValue": "54.3%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-babyvision-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "babyvision",
      "field": "Kimi K3 Max observation",
      "newValue": "85.7%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-charxiv-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "charxiv",
      "field": "Kimi K3 Max observation",
      "newValue": "91.3%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-perceptionbench-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "perceptionbench",
      "field": "Kimi K3 Max observation",
      "newValue": "58.5%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-apex-swe-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "apex-swe",
      "field": "Kimi K3 Max observation",
      "newValue": "48%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-browsecomp-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "browsecomp",
      "field": "Kimi K3 Max observation",
      "newValue": "91.2%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-toolathlon-verified-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "toolathlon-verified",
      "field": "Kimi K3 Max observation",
      "newValue": "76.5%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-mmmu-pro-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "mmmu-pro",
      "field": "Kimi K3 Max observation",
      "newValue": "81.6%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-humanitys-last-exam-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "humanitys-last-exam",
      "field": "Kimi K3 Max observation",
      "newValue": "44.3%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-tau2-bench-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "tau2-bench",
      "field": "Kimi K3 Max observation",
      "newValue": "80.63%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gpqa-diamond-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gpqa-diamond",
      "field": "Kimi K3 Max observation",
      "newValue": "93.5%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-omniscience-index-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-omniscience-index",
      "field": "Kimi K3 Max observation",
      "newValue": "18.42",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-lcr-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-lcr",
      "field": "Kimi K3 Max observation",
      "newValue": "74.7%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-arc-agi-2-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "arc-agi-2",
      "field": "Kimi K3 Max observation",
      "newValue": "60.4%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-chatbot-arena-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "chatbot-arena",
      "field": "Kimi K3 Max observation",
      "newValue": "1489",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livebench-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livebench",
      "field": "Kimi K3 Max observation",
      "newValue": "79.2%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livecodebench-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livecodebench",
      "field": "Kimi K3 Max observation",
      "newValue": "87.19%",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-creative-writing-v3-kimi",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "creative-writing-v3",
      "field": "Kimi K3 Max observation",
      "newValue": "2340.4",
      "reason": "Imported the requested Kimi K3 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-benchmarklist-kimi-k3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-apex-swe-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "apex-swe",
      "field": "Grok 4.6 High observation",
      "newValue": "56.4%",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gpqa-diamond-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gpqa-diamond",
      "field": "Grok 4.6 High observation",
      "newValue": "94.9%",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-omniscience-index-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-omniscience-index",
      "field": "Grok 4.6 High observation",
      "newValue": "30.5",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-lcr-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-lcr",
      "field": "Grok 4.6 High observation",
      "newValue": "75%",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-arc-agi-2-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "arc-agi-2",
      "field": "Grok 4.6 High observation",
      "newValue": "67.1%",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-chatbot-arena-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "chatbot-arena",
      "field": "Grok 4.6 High observation",
      "newValue": "1464",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-arena",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livebench-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livebench",
      "field": "Grok 4.6 High observation",
      "newValue": "78%",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livecodebench-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livecodebench",
      "field": "Grok 4.6 High observation",
      "newValue": "88.22%",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-humanitys-last-exam-grok-4-6-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "humanitys-last-exam",
      "field": "Grok 4.6 High observation",
      "newValue": "42.9%",
      "reason": "Imported the requested Grok 4.6 High value while preserving snapshot, derived-value, and source caveats.",
      "citationId": "citation-grok-4-6-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gdpval-aa-v2-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gdpval-aa-v2",
      "field": "Qwen3.8 Max observation",
      "newValue": "1737",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-briefcase-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-briefcase",
      "field": "Qwen3.8 Max observation",
      "newValue": "1420",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-intelligence-index-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-intelligence-index",
      "field": "Qwen3.8 Max observation",
      "newValue": "58",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-frontiermath-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "frontiermath",
      "field": "Qwen3.8 Max observation",
      "newValue": "46.3%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-deepswe-1-1-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "deepswe-1-1",
      "field": "Qwen3.8 Max observation",
      "newValue": "56.6%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-agents-last-exam-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "agents-last-exam",
      "field": "Qwen3.8 Max observation",
      "newValue": "27%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-snorkel-agents-last-exam",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-harvey-lab-vals-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "harvey-lab-vals",
      "field": "Qwen3.8 Max observation",
      "newValue": "10.4%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-frontierswe-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "frontierswe",
      "field": "Qwen3.8 Max observation",
      "newValue": "73.5%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-jobbench-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "jobbench",
      "field": "Qwen3.8 Max observation",
      "newValue": "53.4%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-babyvision-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "babyvision",
      "field": "Qwen3.8 Max observation",
      "newValue": "91.3%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-charxiv-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "charxiv",
      "field": "Qwen3.8 Max observation",
      "newValue": "93.5%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-perceptionbench-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "perceptionbench",
      "field": "Qwen3.8 Max observation",
      "newValue": "63.5%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-toolathlon-verified-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "toolathlon-verified",
      "field": "Qwen3.8 Max observation",
      "newValue": "72.5%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-mmmu-pro-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "mmmu-pro",
      "field": "Qwen3.8 Max observation",
      "newValue": "82.3%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-humanitys-last-exam-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "humanitys-last-exam",
      "field": "Qwen3.8 Max observation",
      "newValue": "43.6%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gpqa-diamond-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gpqa-diamond",
      "field": "Qwen3.8 Max observation",
      "newValue": "92.6%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-lcr-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-lcr",
      "field": "Qwen3.8 Max observation",
      "newValue": "74.3%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-chatbot-arena-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "chatbot-arena",
      "field": "Qwen3.8 Max observation",
      "newValue": "1491",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livebench-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livebench",
      "field": "Qwen3.8 Max observation",
      "newValue": "78.5%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-livebench",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livecodebench-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livecodebench",
      "field": "Qwen3.8 Max observation",
      "newValue": "87.85%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-swe-bench-pro-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "swe-bench-pro",
      "field": "Qwen3.8 Max observation",
      "newValue": "67.7%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-paperbench-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "paperbench",
      "field": "Qwen3.8 Max observation",
      "newValue": "93%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-qwenreactbench-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "qwenreactbench",
      "field": "Qwen3.8 Max observation",
      "newValue": "1724",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-coworkbench-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "coworkbench",
      "field": "Qwen3.8 Max observation",
      "newValue": "74.8%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-erqa-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "erqa",
      "field": "Qwen3.8 Max observation",
      "newValue": "77.8%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-lvbench-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "lvbench",
      "field": "Qwen3.8 Max observation",
      "newValue": "85.6%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-vision2web-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "vision2web",
      "field": "Qwen3.8 Max observation",
      "newValue": "69%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-mobileworld-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "mobileworld",
      "field": "Qwen3.8 Max observation",
      "newValue": "77.8%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-webarena-verified-qwen-3-8-max-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "webarena-verified",
      "field": "Qwen3.8 Max observation",
      "newValue": "66.8%",
      "reason": "Imported the requested Qwen3.8 Max value while preserving blank cells and source/version caveats.",
      "citationId": "citation-qwen-3-8-benchmark-batch",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gdpval-aa-v2-glm-5-3-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gdpval-aa-v2",
      "field": "GLM-5.3 observation",
      "newValue": "1769",
      "reason": "Imported the requested GLM-5.3 provider value while preserving blank cells and methodology caveats.",
      "citationId": "citation-zai-glm-5-3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-deepswe-1-1-glm-5-3-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "deepswe-1-1",
      "field": "GLM-5.3 observation",
      "newValue": "66.9%",
      "reason": "Imported the requested GLM-5.3 provider value while preserving blank cells and methodology caveats.",
      "citationId": "citation-zai-glm-5-3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-terminal-bench-3-glm-5-3-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "terminal-bench-3",
      "field": "GLM-5.3 observation",
      "newValue": "28.3%",
      "reason": "Imported the requested GLM-5.3 provider value while preserving blank cells and methodology caveats.",
      "citationId": "citation-zai-glm-5-3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-agents-last-exam-glm-5-3-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "agents-last-exam",
      "field": "GLM-5.3 observation",
      "newValue": "28.5%",
      "reason": "Imported the requested GLM-5.3 provider value while preserving blank cells and methodology caveats.",
      "citationId": "citation-zai-glm-5-3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-frontierswe-glm-5-3-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "frontierswe",
      "field": "GLM-5.3 observation",
      "newValue": "78.1%",
      "reason": "Imported the requested GLM-5.3 provider value while preserving blank cells and methodology caveats.",
      "citationId": "citation-zai-glm-5-3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-toolathlon-verified-glm-5-3-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "toolathlon-verified",
      "field": "GLM-5.3 observation",
      "newValue": "73%",
      "reason": "Imported the requested GLM-5.3 provider value while preserving blank cells and methodology caveats.",
      "citationId": "citation-zai-glm-5-3",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gdpval-aa-v2-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gdpval-aa-v2",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "1590",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-intelligence-index-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-intelligence-index",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "53",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-v4-pro-0813",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-deepswe-1-1-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "deepswe-1-1",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "62.7%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-v4-pro-0813",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-harvey-lab-vals-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "harvey-lab-vals",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "7.5%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-vals",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-swe-bench-pro-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "swe-bench-pro",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "55.4%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-v4-pro-0813",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-agents-last-exam-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "agents-last-exam",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "25.7%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-v4-pro-0813",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-browsecomp-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "browsecomp",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "83.4%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-v4-pro-0813",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-toolathlon-verified-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "toolathlon-verified",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "74.1%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-v4-pro-0813",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-humanitys-last-exam-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "humanitys-last-exam",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "41%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gpqa-diamond-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gpqa-diamond",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "92.8%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-lcr-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-lcr",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "75.3%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-chatbot-arena-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "chatbot-arena",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "1465",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-arena",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livecodebench-deepseek-v4-pro-max",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livecodebench",
      "field": "DeepSeek V4 Pro Max observation",
      "newValue": "87.53%",
      "reason": "Imported the requested DeepSeek V4 Pro 0813 value while preserving evaluator, harness, and source caveats.",
      "citationId": "citation-deepseek-vals",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gdpval-aa-v2-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gdpval-aa-v2",
      "field": "Muse Spark 1.2 observation",
      "newValue": "1628",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-briefcase-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-briefcase",
      "field": "Muse Spark 1.2 observation",
      "newValue": "1358",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-intelligence-index-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-intelligence-index",
      "field": "Muse Spark 1.2 observation",
      "newValue": "57",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-deepswe-1-1-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "deepswe-1-1",
      "field": "Muse Spark 1.2 observation",
      "newValue": "59.3%",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-code",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-harvey-lab-vals-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "harvey-lab-vals",
      "field": "Muse Spark 1.2 observation",
      "newValue": "25.42%",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-vals",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-humanitys-last-exam-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "humanitys-last-exam",
      "field": "Muse Spark 1.2 observation",
      "newValue": "45.5%",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gpqa-diamond-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gpqa-diamond",
      "field": "Muse Spark 1.2 observation",
      "newValue": "90.4%",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-omniscience-index-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-omniscience-index",
      "field": "Muse Spark 1.2 observation",
      "newValue": "27.2",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-lcr-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-lcr",
      "field": "Muse Spark 1.2 observation",
      "newValue": "83.3%",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-chatbot-arena-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "chatbot-arena",
      "field": "Muse Spark 1.2 observation",
      "newValue": "1499",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-arena",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livebench-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livebench",
      "field": "Muse Spark 1.2 observation",
      "newValue": "78%",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-livebench",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-toolathlon-verified-muse-spark-1-2-xhigh",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "toolathlon-verified",
      "field": "Muse Spark 1.2 observation",
      "newValue": "75.9%",
      "reason": "Imported the requested Muse Spark 1.2 value while preserving blank cells, Muse Code harness caveats, and Vals Index snapshot caveats.",
      "citationId": "citation-muse-spark-1-2-toolathlon",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gdpval-aa-v2-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gdpval-aa-v2",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "1525",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-briefcase-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-briefcase",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "1132",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-intelligence-index-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-intelligence-index",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "56",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-cursorbench-3-2-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "cursorbench-3-2",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "61.6%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-cursorbench-32",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-deepswe-1-1-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "deepswe-1-1",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "65.3%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-google",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-terminal-bench-3-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "terminal-bench-3",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "14.9%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-google",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-harvey-lab-vals-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "harvey-lab-vals",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "8.75%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-vals",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-agents-last-exam-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "agents-last-exam",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "26.3%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-google",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-charxiv-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "charxiv",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "88.7%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-google",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-lvbench-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "lvbench",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "85.4%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-google",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-humanitys-last-exam-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "humanitys-last-exam",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "47.9%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-gpqa-diamond-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "gpqa-diamond",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "94.5%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-omniscience-index-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-omniscience-index",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "26.5",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-aa-lcr-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "aa-lcr",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "80%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-aa",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-chatbot-arena-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "chatbot-arena",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "1490",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-arena",
      "actor": "benchmark-sheet-ingestion"
    },
    {
      "id": "revision-ingest-sheet-livebench-gemini-3-7-flash-high",
      "revisedAt": "2026-08-16T00:00:00.000Z",
      "changeType": "corrected",
      "benchmarkId": "livebench",
      "field": "Gemini 3.7 Flash observation",
      "newValue": "78.8%",
      "reason": "Imported the requested Gemini 3.7 Flash high-reasoning value while preserving protocol, evaluator, and Vals Index version caveats.",
      "citationId": "citation-gemini-3-7-flash-livebench",
      "actor": "benchmark-sheet-ingestion"
    }
  ],
  "frontierCohort": {
    "modelIds": [
      "claude-opus-5-max",
      "claude-fable-5-max",
      "gpt-5-6-sol-max",
      "grok-4-6-xhigh",
      "gemini-3-1-pro-high",
      "kimi-k3-max",
      "deepseek-v4-pro-max"
    ],
    "entries": [
      {
        "modelId": "claude-opus-5-max",
        "requiredEffort": "max",
        "allowDefaultEffort": false
      },
      {
        "modelId": "claude-fable-5-max",
        "requiredEffort": "max",
        "allowDefaultEffort": false
      },
      {
        "modelId": "gpt-5-6-sol-max",
        "requiredEffort": "max",
        "allowDefaultEffort": false
      },
      {
        "modelId": "grok-4-6-xhigh",
        "requiredEffort": "xhigh",
        "allowDefaultEffort": false
      },
      {
        "modelId": "gemini-3-1-pro-high",
        "requiredEffort": "high",
        "allowDefaultEffort": true
      },
      {
        "modelId": "kimi-k3-max",
        "requiredEffort": "max",
        "allowDefaultEffort": true
      },
      {
        "modelId": "deepseek-v4-pro-max",
        "requiredEffort": "max",
        "allowDefaultEffort": false
      }
    ],
    "maxMissing": 2,
    "maxAgeDays": 90
  }
}
